Xunzhuo commited on
Commit
a2af77f
·
verified ·
1 Parent(s): d23722a

Add System One typed batch API and matching fine-tuning inputs

Browse files
FINETUNING.md CHANGED
@@ -1,5 +1,7 @@
1
  # Fine-tune on your data
2
 
 
 
3
  The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
4
 
5
  Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
 
1
  # Fine-tune on your data
2
 
3
+ Already using System One requests? [Convert the same state/questions/criteria into training rows](SYSTEM_ONE.md#fine-tune-with-the-same-inputs) with separate hard or soft targets and shared source-component IDs.
4
+
5
  The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
6
 
7
  Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
PACKAGE_MANIFEST.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
- "bytes_excluding_this_manifest": 2327383853,
3
- "file_count_excluding_this_manifest": 96,
4
  "files": {
5
  ".gitattributes": {
6
  "bytes": 753,
@@ -23,8 +23,8 @@
23
  "sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
24
  },
25
  "FINETUNING.md": {
26
- "bytes": 3629,
27
- "sha256": "33c76f3490f283d9e189eeb4190c0297795329b56546271a52b80cd503dd2ec0"
28
  },
29
  "LICENSE": {
30
  "bytes": 11358,
@@ -75,13 +75,21 @@
75
  "sha256": "7e275e73c18b756a20b886272750b58416ead2c345b12393e162050231aa8f64"
76
  },
77
  "README.md": {
78
- "bytes": 4184,
79
- "sha256": "ac04717649be2e2e35355e882cb867c675c684651bccd8ed33235ecf3ae0e9c8"
80
  },
81
  "RUNTIME_UPDATE.json": {
82
  "bytes": 1087,
83
  "sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
84
  },
 
 
 
 
 
 
 
 
85
  "TECHNICAL_VALIDATION.json": {
86
  "bytes": 1547,
87
  "sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
@@ -95,8 +103,8 @@
95
  "sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
96
  },
97
  "USAGE.md": {
98
- "bytes": 3142,
99
- "sha256": "13205f6361ce04807b4de0a13c3b6128c0b89e81f2422177d7a04c4b19a67f8e"
100
  },
101
  "VALIDATION.md": {
102
  "bytes": 2120,
@@ -182,9 +190,13 @@
182
  "bytes": 5751,
183
  "sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
184
  },
 
 
 
 
185
  "decision_inference/__init__.py": {
186
- "bytes": 193,
187
- "sha256": "198b3c64fd20764d3d53489306cb16da9ecb04e6e98a002c5a6ee42f18707f87"
188
  },
189
  "decision_inference/_auto.py": {
190
  "bytes": 1325,
@@ -194,6 +206,10 @@
194
  "bytes": 3026,
195
  "sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
196
  },
 
 
 
 
197
  "decision_inference/profile.py": {
198
  "bytes": 2071,
199
  "sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
@@ -222,6 +238,10 @@
222
  "bytes": 1242,
223
  "sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
224
  },
 
 
 
 
225
  "infer.py": {
226
  "bytes": 2373,
227
  "sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
@@ -385,6 +405,10 @@
385
  "source-metadata/VELA_LICENSE_SOURCE_README.md": {
386
  "bytes": 2046,
387
  "sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
 
 
 
 
388
  }
389
  },
390
  "language_scope": "English typed-decisions specialist",
@@ -402,6 +426,7 @@
402
  },
403
  "parameter_tensors": 489,
404
  "parameters": 571909635,
 
405
  "publication": {
406
  "download_access": "public_ungated",
407
  "new_contribution_license": "Apache-2.0",
@@ -415,6 +440,20 @@
415
  "schema": "decision.public-distribution.v1",
416
  "status": "READY_FOR_ROOT_PUBLICATION",
417
  "subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
418
  "technical_validation": "TECHNICAL_VALIDATION.json",
419
  "training_provenance": "TRAINING_PROVENANCE.json"
420
  }
 
1
  {
2
+ "bytes_excluding_this_manifest": 2327409686,
3
+ "file_count_excluding_this_manifest": 102,
4
  "files": {
5
  ".gitattributes": {
6
  "bytes": 753,
 
23
  "sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
24
  },
25
  "FINETUNING.md": {
26
+ "bytes": 3842,
27
+ "sha256": "2874eedec3b19462552a35e0dc99af29c456b3a2f8096cc61344b92e4cc6db41"
28
  },
29
  "LICENSE": {
30
  "bytes": 11358,
 
75
  "sha256": "7e275e73c18b756a20b886272750b58416ead2c345b12393e162050231aa8f64"
76
  },
77
  "README.md": {
78
+ "bytes": 4416,
79
+ "sha256": "dd76e4f9879defc462441f899974140e9a65a6361f26dff883fcedfc473381af"
80
  },
81
  "RUNTIME_UPDATE.json": {
82
  "bytes": 1087,
83
  "sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
84
  },
85
+ "SYSTEM_ONE.md": {
86
+ "bytes": 5735,
87
+ "sha256": "414e12680a6ec2dbbbc660d1f6ba0d601fa941388476a62696b7f5e8a3e11ef9"
88
+ },
89
+ "SYSTEM_ONE_VALIDATION.json": {
90
+ "bytes": 2583,
91
+ "sha256": "24c28a2b8c448507f26e116a7f6a6d4e40fb8eaae0384e6bee421a105e415e0a"
92
+ },
93
  "TECHNICAL_VALIDATION.json": {
94
  "bytes": 1547,
95
  "sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
 
103
  "sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
104
  },
105
  "USAGE.md": {
106
+ "bytes": 4589,
107
+ "sha256": "581e65562cdf35d8eac544640b984e6267d86e3b46adcaf1a00bd9d9c5af247c"
108
  },
109
  "VALIDATION.md": {
110
  "bytes": 2120,
 
190
  "bytes": 5751,
191
  "sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
192
  },
193
+ "decision_finetune/system_one.py": {
194
+ "bytes": 1784,
195
+ "sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
196
+ },
197
  "decision_inference/__init__.py": {
198
+ "bytes": 282,
199
+ "sha256": "8199829090f1a3ca5fe59aec1a6dce2beb57026c301d2c87d8c5b1c960981e66"
200
  },
201
  "decision_inference/_auto.py": {
202
  "bytes": 1325,
 
206
  "bytes": 3026,
207
  "sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
208
  },
209
+ "decision_inference/_system_one.py": {
210
+ "bytes": 10690,
211
+ "sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
212
+ },
213
  "decision_inference/profile.py": {
214
  "bytes": 2071,
215
  "sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
 
238
  "bytes": 1242,
239
  "sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
240
  },
241
+ "examples/system-one.json": {
242
+ "bytes": 692,
243
+ "sha256": "0b8320ecbeded8c3229dccc52f5045f00e8d1968183bc9d1371d24e949925dcf"
244
+ },
245
  "infer.py": {
246
  "bytes": 2373,
247
  "sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
 
405
  "source-metadata/VELA_LICENSE_SOURCE_README.md": {
406
  "bytes": 2046,
407
  "sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
408
+ },
409
+ "systemone.py": {
410
+ "bytes": 2368,
411
+ "sha256": "8ce311c5651f1b0c1bc99d5a4ccaa6c7878b5c509256e4668a2191ad8c35e2e3"
412
  }
413
  },
414
  "language_scope": "English typed-decisions specialist",
 
426
  },
427
  "parameter_tensors": 489,
428
  "parameters": 571909635,
429
+ "public_non_native_bytes_excluding_this_manifest": 5150475,
430
  "publication": {
431
  "download_access": "public_ungated",
432
  "new_contribution_license": "Apache-2.0",
 
440
  "schema": "decision.public-distribution.v1",
441
  "status": "READY_FOR_ROOT_PUBLICATION",
442
  "subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
443
+ "system_one_api": {
444
+ "default_physical_batch_limit": 8,
445
+ "entry": "decision_inference.SystemOne",
446
+ "finetune_conversion": "decision_finetune.system_one.system_one_training_rows",
447
+ "max_batch_decisions": 512,
448
+ "max_questions": 128,
449
+ "max_request_bytes": 2097152,
450
+ "max_requests": 128,
451
+ "optional_auto_capacity": 32,
452
+ "request": "state/model/questions",
453
+ "response": "model/answers/usage",
454
+ "validation": "SYSTEM_ONE_VALIDATION.json",
455
+ "weights_unchanged": true
456
+ },
457
  "technical_validation": "TECHNICAL_VALIDATION.json",
458
  "training_provenance": "TRAINING_PROVENANCE.json"
459
  }
README.md CHANGED
@@ -54,9 +54,11 @@ Lex is evaluated as an English specialist on these four workflows. For broader m
54
 
55
  ## Make it yours
56
 
 
 
57
  Run the [Python examples](USAGE.md), or adapt Lex to your own labels and rubrics with the included [fine-tuning CLI](FINETUNING.md). Both hard and soft training labels are supported, with checkpoint resume.
58
 
59
- The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the Studio adapter uses 2–10 Score levels.
60
 
61
  ## Architecture
62
 
 
54
 
55
  ## Make it yours
56
 
57
+ **One state. Many decisions.** Use the [System One API](SYSTEM_ONE.md) to submit up to 128 typed questions, or batch the same questions across independent contexts. Results return under your original question IDs.
58
+
59
  Run the [Python examples](USAGE.md), or adapt Lex to your own labels and rubrics with the included [fine-tuning CLI](FINETUNING.md). Both hard and soft training labels are supported, with checkpoint resume.
60
 
61
+ The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
62
 
63
  ## Architecture
64
 
SYSTEM_ONE.md ADDED
@@ -0,0 +1,129 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # One state. Many decisions.
2
+
3
+ Decision uses the [System One API format](https://docs.typesafe.ai/api): supply
4
+ `state`, `model` and a map of typed `questions`; receive an `answers` map under
5
+ the same question IDs. Noul returns the probability of yes, Choice selects from
6
+ your named options, and Score evaluates an ordered rubric.
7
+
8
+ ```python
9
+ from decision_inference import SystemOne
10
+
11
+ # native is your loaded Decision checkpoint; see USAGE.md for loading.
12
+ client = SystemOne(native, batching="auto")
13
+ result = client.system_one(
14
+ state={"message": "Please refund the duplicate charge. I need this fixed today."},
15
+ questions={
16
+ "refund_requested": {
17
+ "type": "noul",
18
+ "instructions": "Does the customer explicitly request a refund?",
19
+ },
20
+ "team": {
21
+ "type": "choice",
22
+ "instructions": "Which team should handle this request?",
23
+ "criteria": {"Billing": "Charges and refunds", "Support": "Technical problems"},
24
+ },
25
+ "urgency": {
26
+ "type": "score",
27
+ "instructions": "How urgent is the request?",
28
+ "criteria": ["No deadline", "Needed soon", "Needed today"],
29
+ },
30
+ },
31
+ )
32
+ print(result["answers"])
33
+ ```
34
+
35
+ `client.evaluate(request)` accepts the complete JSON request object, including
36
+ `model`. A client is bound to its loaded checkpoint; a mismatched model is an
37
+ error. Public Kai/Lex names are inferred from their verified native manifest.
38
+ For your own fine-tune, use `SystemOne(native, model="my-decision-model")`.
39
+
40
+ ## Batch the questions or the contexts
41
+
42
+ One request accepts **up to 128 questions**. For one set of questions across many
43
+ independent contexts, use `client.batch`:
44
+
45
+ ```python
46
+ questions = {
47
+ "refund_requested": {
48
+ "type": "noul",
49
+ "instructions": "Does the customer explicitly request a refund?",
50
+ }
51
+ }
52
+ results = client.batch([
53
+ {"model": client.model, "state": message, "questions": questions}
54
+ for message in messages
55
+ ])
56
+ # results[i] corresponds to messages[i]; question IDs can repeat across requests.
57
+ ```
58
+
59
+ `batch` is a Decision Python extension around ordinary System One requests. It
60
+ accepts up to 128 requests and 512 total decisions, with a combined 2 MiB input
61
+ limit. Responses preserve request order and question order. The entire batch
62
+ must pass validation and complete-input token admission before any forward.
63
+ An overlength or malformed input fails the call without partial answers.
64
+
65
+ The default uses physical batches of up to eight. `batching="auto"` uses the
66
+ released padding-aware scheduler: up to 32 consecutive same-type decisions when
67
+ that adds no padding; otherwise it keeps B8. FP32 rounding can vary with physical
68
+ batch shape. Each state/question pair still has its own encoder computation.
69
+ One API call does not imply one forward or a shared state activation cache.
70
+
71
+ ## Typed fields
72
+
73
+ | Type | `criteria` | Answer |
74
+ |---|---|---|
75
+ | `noul` | Optional `true` / `false` descriptions | `type`, `noul` |
76
+ | `choice` | 2–255 named options; descriptions may be null | `type`, `choice`, `probabilities`, `confidence` |
77
+ | `score` | 2–10 ordered level descriptions | `type`, `score`, `legend`, `probabilities`, `confidence` |
78
+
79
+ State, instructions and descriptions accept strings, JSON objects or arrays.
80
+ Structured values become deterministic compact JSON with sorted object keys;
81
+ array order, Choice option order and Score level order are preserved. Question
82
+ IDs are bookkeeping only. Choice names are part of the semantic input, including
83
+ when their description is null. Score levels are indexed from zero; `score`
84
+ preserves the native probability-weighted FP32 expectation. Structured Score
85
+ descriptions appear as JSON strings in `legend`.
86
+
87
+ Every complete state/question/candidate sequence must fit **1,024 tokens**.
88
+ Nothing is truncated. The token count includes the repeated state for each
89
+ question; `usage.input_tokens` sums these complete sequences and
90
+ `usage.output_tokens` is zero because the model returns scores without generating
91
+ text. These are local computation counts, not TypeSafe billing counts.
92
+
93
+ Decision's `confidence` is the largest candidate probability, matching its native
94
+ runtime. It is not calibrated correctness. TypeSafe does not specify its own
95
+ formula in the [confidence documentation](https://docs.typesafe.ai/confidence),
96
+ so thresholds should not be transferred between models without validation.
97
+ The shared request/answer schema does not claim identical weights, confidence
98
+ values, hosted service limits or SDK behavior.
99
+
100
+ ## Fine-tune with the same inputs
101
+
102
+ Use the same conversion for training so structured inputs, option names and
103
+ criteria have identical semantics at training and inference time:
104
+
105
+ ```python
106
+ import json
107
+ from pathlib import Path
108
+ from decision_finetune.system_one import system_one_training_rows
109
+
110
+ request = json.loads(Path("examples/system-one.json").read_text())
111
+ rows = system_one_training_rows(
112
+ request, # the same model/state/questions object used for inference
113
+ targets={
114
+ "refund_requested": {"probability": 1.0},
115
+ "team": {"choice_id": "Billing"},
116
+ "urgency": {"probabilities": [0.0, 0.0, 1.0]},
117
+ },
118
+ request_id="ticket-1001",
119
+ source_id="support-tickets",
120
+ component_id="customer-42",
121
+ )
122
+ ```
123
+
124
+ Save the rows as JSONL for the existing [fine-tuning CLI](FINETUNING.md). Targets
125
+ stay separate from state, instructions and criteria. Related examples share a
126
+ source/component ID and stay in one split; the CLI checks component and exact
127
+ input overlap. Optional `hard_target_ids` supplies separate evaluation labels
128
+ under the same question IDs. Noul supports soft yes probabilities; Choice and
129
+ Score support complete soft distributions in the original candidate order.
SYSTEM_ONE_VALIDATION.json ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "all_reference_outputs_exact": true,
3
+ "api_source": {
4
+ "bytes": 10690,
5
+ "sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
6
+ },
7
+ "complete_input_tokens": 1024,
8
+ "coverage": [
9
+ "Noul, Choice and Score",
10
+ "128 questions",
11
+ "128 contexts and 512 decisions",
12
+ "question rename and reorder",
13
+ "irrelevant question insertion",
14
+ "single and multi-context agreement",
15
+ "default B8 and opt-in padding-aware B32",
16
+ "late invalid input before any forward",
17
+ "same training and inference input encoding"
18
+ ],
19
+ "cross_batch_tolerance": 2e-05,
20
+ "device": "AMD GPU / ROCm",
21
+ "limits": "Technical compatibility validation on synthetic unlabeled cases; not evidence of accuracy, calibration or latency gains.",
22
+ "max_decisions_per_batch": 512,
23
+ "max_questions_per_request": 128,
24
+ "max_requests_per_batch": 128,
25
+ "models": {
26
+ "Kai": {
27
+ "all_489_parameters_and_66_buffers_unchanged": true,
28
+ "cross_case_checks": 713,
29
+ "forward_calls": 242,
30
+ "forward_rows": 2756,
31
+ "hard_flips": 0,
32
+ "invalid_input_checks": 7,
33
+ "invalid_input_forward_calls": 0,
34
+ "max_probability_error": 2.0563602447509766e-06,
35
+ "max_score_error": 2.980232238769531e-07,
36
+ "native_manifest_sha256": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
37
+ "result_sha256": "8cb3bc435e36b79d96b60058e9b4c57b9996dcb8a55a1cd3faf6c7324e7f917c",
38
+ "same_physical_reference_outputs_exact": true
39
+ },
40
+ "Lex": {
41
+ "all_489_parameters_and_66_buffers_unchanged": true,
42
+ "cross_case_checks": 713,
43
+ "forward_calls": 242,
44
+ "forward_rows": 2756,
45
+ "hard_flips": 0,
46
+ "invalid_input_checks": 7,
47
+ "invalid_input_forward_calls": 0,
48
+ "max_probability_error": 1.1324882507324219e-06,
49
+ "max_score_error": 7.152557373046875e-07,
50
+ "native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
51
+ "result_sha256": "e94c5f1109f15005415a48623602d5255b38bd33451b55c758a0db0c16f68191",
52
+ "same_physical_reference_outputs_exact": true
53
+ }
54
+ },
55
+ "optimizer_updates": 0,
56
+ "performance_benchmark": false,
57
+ "quality_evaluation": false,
58
+ "reference": "Independent System One mapping using the original Studio single-question conversion and native predictor; identical physical batches.",
59
+ "status": "PASS_AMD_SYSTEM_ONE_API",
60
+ "total_forward_calls": 484,
61
+ "training_conversion_source": {
62
+ "bytes": 1784,
63
+ "sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
64
+ },
65
+ "weights_unchanged": true
66
+ }
USAGE.md CHANGED
@@ -7,12 +7,53 @@ hf download llm-semantic-router/Decision-1.0-Lex --local-dir Decision-1.0-Lex
7
  cd Decision-1.0-Lex
8
  ```
9
 
10
- Use `--local-dir` to materialize ordinary files for the native loader. Install the pinned Python dependencies from `requirements.txt` in your compatible ROCm environment before running the examples.
11
 
12
  This initial release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
13
 
14
  Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
15
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  ```bash
17
  export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
18
  export PYTHONDONTWRITEBYTECODE=1
 
7
  cd Decision-1.0-Lex
8
  ```
9
 
10
+ Use `--local-dir` to materialize ordinary files for the native loader. Use the compatible ROCm/Python environment described below.
11
 
12
  This initial release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
13
 
14
  Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
15
 
16
+ ## System One: parallel typed questions
17
+
18
+ Use the standard `state / model / questions` request shape. The response contains
19
+ `model`, an `answers` map under the same question IDs, and `usage`.
20
+
21
+ ```bash
22
+ ROCR_VISIBLE_DEVICES=0 python systemone.py \
23
+ --input examples/system-one.json --output answers.json --batching auto
24
+ ```
25
+
26
+ ```python
27
+ from decision_runtime import load_native
28
+ from decision_inference import SystemOne
29
+
30
+ native = load_native(
31
+ "native",
32
+ expected_manifest_sha256="f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
33
+ device="cuda:0",
34
+ )
35
+ client = SystemOne(native, batching="auto")
36
+ answers = client.system_one(
37
+ state="Please refund the duplicate charge. I need this fixed today.",
38
+ questions={
39
+ "refund": {"type": "noul", "instructions": "Is a refund requested?"},
40
+ "urgent": {"type": "noul", "instructions": "Does the customer specify a deadline?"},
41
+ },
42
+ )["answers"]
43
+ ```
44
+
45
+ Up to 128 questions per request. `client.batch(requests)` combines independent
46
+ contexts with up to 512 total decisions. Each complete state/question/candidate
47
+ sequence must fit 1,024 tokens; validation happens before the first forward.
48
+ Default physical batching is B8; `auto` opts into the released homogeneous,
49
+ padding-aware B32 scheduler. [Fields, batching and the matching fine-tuning
50
+ conversion](SYSTEM_ONE.md).
51
+
52
+ ## Native records
53
+
54
+ The original native JSONL interface remains available for applications that
55
+ need logits, explicit candidate IDs or arbitrary ordered Score values.
56
+
57
  ```bash
58
  export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
59
  export PYTHONDONTWRITEBYTECODE=1
decision_finetune/system_one.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Keep System One inference and fine-tuning input semantics identical."""
2
+ import copy
3
+ import json
4
+
5
+
6
+ def system_one_training_rows(request, targets, *, request_id, source_id,
7
+ component_id, hard_target_ids=None):
8
+ """Convert a typed request plus separate native targets to training rows.
9
+
10
+ targets is keyed by question ID. Noul uses {'probability': p_yes}, Choice
11
+ {'choice_id': label} or {'probabilities': [...]}, Score {'probabilities': [...]}.
12
+ The caller assigns shared source/component IDs across related examples so
13
+ existing TRAIN/DEV validation can reject component overlap.
14
+ """
15
+ from decision_inference._system_one import system_one_records
16
+ from .data import semantics
17
+ for value in (request_id, source_id, component_id):
18
+ if not isinstance(value, str) or not value.strip():
19
+ raise ValueError("Explicit nonempty request/source/component IDs required")
20
+ rows = system_one_records(request)
21
+ qids = {row["question"]["id"] for row in rows}
22
+ if not isinstance(targets, dict) or set(targets) != qids:
23
+ raise ValueError("Exactly one target per question ID is required")
24
+ if hard_target_ids is not None and (not isinstance(hard_target_ids, dict) or set(hard_target_ids) != qids):
25
+ raise ValueError("Hard evaluation labels must cover exactly the question IDs")
26
+ for i, row in enumerate(rows):
27
+ qid = row["question"]["id"]
28
+ row.update(id=json.dumps([request_id, i], ensure_ascii=False, separators=(",", ":")),
29
+ source_id=source_id, component_id=component_id,
30
+ target=copy.deepcopy(targets[qid]))
31
+ if hard_target_ids is not None:
32
+ row["hard_target_id"] = hard_target_ids[qid]
33
+ semantics(row)
34
+ return rows
decision_inference/__init__.py CHANGED
@@ -1,5 +1,5 @@
1
  from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
2
-
3
  from ._auto import predict_auto_1k
 
4
 
5
- __all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k"]
 
1
  from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
 
2
  from ._auto import predict_auto_1k
3
+ from ._system_one import SystemOne, system_one_records
4
 
5
+ __all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k", "SystemOne", "system_one_records"]
decision_inference/_system_one.py ADDED
@@ -0,0 +1,217 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """System One request/answer schema over complete-input Decision inference.
2
+
3
+ Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping;
4
+ Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache.
5
+ """
6
+ import copy
7
+ import json
8
+ import math
9
+
10
+ MAX_QUESTIONS = 128
11
+ MAX_REQUESTS = 128
12
+ MAX_DECISIONS = 512
13
+ MAX_REQUEST_BYTES = 2 * 1024 * 1024
14
+ PUBLIC_MODELS = {
15
+ "Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
16
+ "Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
17
+ }
18
+
19
+
20
+ def _identifier(value, label):
21
+ if not isinstance(value, str) or not value.strip() or len(value) > 128:
22
+ raise ValueError(label + " must be a nonempty string of at most 128 characters")
23
+ return value
24
+
25
+
26
+ def _json(value):
27
+ # JSON objects must have string keys: never silently coerce Python keys.
28
+ def check(item):
29
+ if isinstance(item, dict):
30
+ if not all(isinstance(k, str) for k in item):
31
+ raise ValueError("JSON object keys must be strings")
32
+ for v in item.values():
33
+ check(v)
34
+ elif isinstance(item, list):
35
+ for v in item:
36
+ check(v)
37
+ elif item is not None and not isinstance(item, (str, bool, int, float)):
38
+ raise ValueError("Only JSON values are supported")
39
+ try:
40
+ check(value)
41
+ return json.dumps(value, ensure_ascii=False, sort_keys=True,
42
+ separators=(",", ":"), allow_nan=False)
43
+ except (TypeError, RecursionError, UnicodeError) as exc:
44
+ raise ValueError("Invalid JSON content") from exc
45
+
46
+
47
+ def _content(value, label):
48
+ if isinstance(value, str):
49
+ if not value.strip():
50
+ raise ValueError(label + " must not be empty")
51
+ return value
52
+ if isinstance(value, (dict, list)):
53
+ return _json(value)
54
+ raise ValueError(label + " must be text, an object, or an array")
55
+
56
+
57
+ def system_one_records(request):
58
+ """Validate one wire request and return native rows, without loading a model.
59
+
60
+ Full token admission occurs in predict_1k before the first model forward.
61
+ External record IDs should be made unique when combining training examples.
62
+ """
63
+ if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}:
64
+ raise ValueError("A request contains exactly model, state and questions")
65
+ _identifier(request["model"], "Model")
66
+ if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES:
67
+ raise ValueError("Request exceeds 2 MiB; no input is truncated")
68
+ state = _content(request["state"], "State")
69
+ questions = request["questions"]
70
+ if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS:
71
+ raise ValueError("Provide 1..128 named questions")
72
+ rows = []
73
+ for index, (qid, item) in enumerate(questions.items()):
74
+ _identifier(qid, "Question ID")
75
+ if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"}
76
+ or not {"type", "instructions"} <= set(item)):
77
+ raise ValueError(qid + ": use type, instructions and optional criteria")
78
+ kind = item["type"]
79
+ if kind not in ("noul", "choice", "score"):
80
+ raise ValueError(qid + ": type must be noul, choice or score")
81
+ q = {"id": qid, "type": kind.capitalize(),
82
+ "text": _content(item["instructions"], qid + ".instructions")}
83
+ criteria = item.get("criteria")
84
+ if kind == "choice":
85
+ if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255:
86
+ raise ValueError(qid + ": Choice requires 2..255 named options")
87
+ q["options"] = []
88
+ for name, description in criteria.items():
89
+ _identifier(name, "Choice option")
90
+ text = name if description is None else name + ": " + _content(description, qid + ".criteria")
91
+ q["options"].append({"id": name, "text": text})
92
+ elif kind == "score":
93
+ if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
94
+ raise ValueError(qid + ": Score requires 2..10 ordered levels")
95
+ q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")}
96
+ for i, v in enumerate(criteria)]
97
+ elif "criteria" in item:
98
+ if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
99
+ raise ValueError(qid + ": Noul criteria accept false and true only")
100
+ for key in ("false", "true"):
101
+ if key in criteria:
102
+ q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key)
103
+ rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q})
104
+ return rows
105
+
106
+
107
+ def _answer(row, prediction):
108
+ q = row["question"]
109
+ kind = q["type"].lower()
110
+ ids = (["no", "yes"] if kind == "noul" else
111
+ [v["id"] for v in q["options" if kind == "choice" else "levels"]])
112
+ if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"]
113
+ or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids
114
+ or type(prediction.get("input_tokens")) is not int
115
+ or not 1 <= prediction["input_tokens"] <= 1024
116
+ or type(prediction.get("state_tokens_original")) is not int
117
+ or prediction["state_tokens_original"] < 0
118
+ or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")):
119
+ raise RuntimeError("Prediction identity or complete-input profile mismatch")
120
+ p = prediction.get("probabilities")
121
+ if (not isinstance(p, list) or len(p) != len(ids)
122
+ or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p)
123
+ or abs(sum(p) - 1) > 2e-5):
124
+ raise RuntimeError("Invalid prediction probabilities")
125
+ answer = {"type": kind}
126
+ if kind == "noul":
127
+ if prediction.get("probability") != p[1]:
128
+ raise RuntimeError("Native Noul probability mismatch")
129
+ answer["noul"] = p[1]
130
+ return answer
131
+ best = ids[max(range(len(p)), key=p.__getitem__)]
132
+ if prediction.get("choice_id") != best or prediction.get("confidence") != max(p):
133
+ raise RuntimeError("Native Choice/confidence mismatch")
134
+ answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"])
135
+ if kind == "choice":
136
+ answer["choice"] = best
137
+ else:
138
+ score = prediction.get("score")
139
+ if (type(score) not in (int, float) or not math.isfinite(score)
140
+ or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5):
141
+ raise RuntimeError("Native ordinal Score mismatch")
142
+ # Preserve native FP32 arithmetic, not a new CPU reduction.
143
+ answer["score"] = score
144
+ answer["legend"] = {v["id"]: v["text"] for v in q["levels"]}
145
+ return answer
146
+
147
+
148
+ class SystemOne:
149
+ """Local System One API for a loaded Kai, Lex or compatible fine-tune.
150
+
151
+ evaluate(request) accepts the HTTP body shape; system_one(**request) is its
152
+ Python equivalent. batch(requests) flattens independent states into GPU
153
+ batches and restores the original request/question order. B8 is the default;
154
+ batching='auto' opts into the published homogeneous padding-aware B32 path.
155
+ """
156
+ def __init__(self, native, *, model=None, batching="default"):
157
+ if model is None:
158
+ model = next((name for name, sha in PUBLIC_MODELS.items()
159
+ if sha == native.manifest_sha256), None)
160
+ _identifier(model, "Model (required for a custom fine-tune)")
161
+ # Do not let a different loaded checkpoint claim a published identity.
162
+ if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]:
163
+ raise ValueError("Loaded checkpoint does not match the public model name")
164
+ if batching not in ("default", "auto"):
165
+ raise ValueError("batching must be default or auto")
166
+ self.native, self.model, self.batching = native, model, batching
167
+
168
+ def system_one(self, *, state, questions, model=None):
169
+ return self.evaluate({"model": self.model if model is None else model,
170
+ "state": state, "questions": questions})
171
+
172
+ def evaluate(self, request):
173
+ return self.batch([request])[0]
174
+
175
+ def batch(self, requests):
176
+ if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS:
177
+ raise ValueError("Provide 1..128 request objects")
178
+ if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES:
179
+ raise ValueError("Combined request exceeds 2 MiB")
180
+ # Detach mutable caller inputs before conversion/admission/inference.
181
+ requests = copy.deepcopy(requests)
182
+ groups = []
183
+ for request in requests:
184
+ rows = system_one_records(request)
185
+ if request["model"] != self.model:
186
+ raise ValueError("Request model does not match this loaded model")
187
+ groups.append(rows)
188
+ count = sum(map(len, groups))
189
+ if count > MAX_DECISIONS:
190
+ raise ValueError("Provide at most 512 decisions in one batch")
191
+ records, slots = [], []
192
+ # Question-major order permits the same question across many states to
193
+ # share a physical batch; external IDs never decide caching or grouping.
194
+ for qi in range(max(map(len, groups))):
195
+ for ri, group in enumerate(groups):
196
+ if qi < len(group):
197
+ row = group[qi]
198
+ row["id"] = f"systemone:{ri}:{qi}"
199
+ records.append(row)
200
+ slots.append((ri, qi))
201
+ if self.batching == "auto":
202
+ from ._auto import predict_auto_1k
203
+ predictions = predict_auto_1k(self.native, records)
204
+ else:
205
+ from ._request import predict_1k
206
+ predictions = predict_1k(self.native, records, batch_size=8)
207
+ if len(predictions) != len(records):
208
+ raise RuntimeError("Incomplete model result; no partial answers returned")
209
+ values = [[None] * len(group) for group in groups]
210
+ tokens = [0] * len(groups)
211
+ for row, prediction, (ri, qi) in zip(records, predictions, slots):
212
+ values[ri][qi] = _answer(row, prediction)
213
+ tokens[ri] += prediction["input_tokens"]
214
+ return [{"model": self.model,
215
+ "answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])},
216
+ "usage": {"input_tokens": tokens[ri], "output_tokens": 0}}
217
+ for ri, group in enumerate(groups)]
examples/system-one.json ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "Decision-1.0-Lex",
3
+ "state": {
4
+ "message": "Please refund the duplicate charge. I need this fixed today."
5
+ },
6
+ "questions": {
7
+ "refund_requested": {
8
+ "type": "noul",
9
+ "instructions": "Does the customer explicitly request a refund?"
10
+ },
11
+ "team": {
12
+ "type": "choice",
13
+ "instructions": "Which team should handle this request?",
14
+ "criteria": {
15
+ "Billing": "Charges and refunds",
16
+ "Support": "Technical problems"
17
+ }
18
+ },
19
+ "urgency": {
20
+ "type": "score",
21
+ "instructions": "How urgent is the request?",
22
+ "criteria": [
23
+ "No deadline",
24
+ "Needed soon",
25
+ "Needed today"
26
+ ]
27
+ }
28
+ }
29
+ }
systemone.py ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Run one System One request, or a batch of requests, on a local Decision model."""
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+
6
+ NATIVE_MANIFEST_SHA256 = 'f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6'
7
+
8
+
9
+ def _unique_object(pairs):
10
+ result = {}
11
+ for key, value in pairs:
12
+ if key in result:
13
+ raise ValueError('Duplicate JSON object key: ' + key)
14
+ result[key] = value
15
+ return result
16
+
17
+
18
+ def main():
19
+ parser = argparse.ArgumentParser(description='Decision System One: typed parallel questions on AMD')
20
+ parser.add_argument('--native', default=str(Path(__file__).resolve().parent / 'native'))
21
+ parser.add_argument('--manifest-sha256', default=NATIVE_MANIFEST_SHA256)
22
+ parser.add_argument('--model', help='Explicit name for a compatible custom fine-tune')
23
+ parser.add_argument('--input', required=True, help='JSON request or array of request objects')
24
+ parser.add_argument('--output', required=True, help='Fresh response JSON file')
25
+ parser.add_argument('--batching', choices=('default', 'auto'), default='default')
26
+ args = parser.parse_args()
27
+ payload = Path(args.input).read_bytes()
28
+ if len(payload) > 2 * 1024 * 1024:
29
+ raise ValueError('Input file exceeds 2 MiB')
30
+ request = json.loads(payload, object_pairs_hook=_unique_object)
31
+ if Path(args.output).exists():
32
+ raise ValueError('Output must be fresh')
33
+ import torch
34
+ from decision_runtime import load_native
35
+ from decision_inference import SystemOne
36
+ if torch.version.hip is None or not torch.cuda.is_available() or torch.cuda.device_count() != 1:
37
+ raise RuntimeError('Expose exactly one AMD ROCm GPU')
38
+ torch.cuda.set_device(0)
39
+ torch.set_num_threads(2)
40
+ torch.backends.cuda.matmul.allow_tf32 = False
41
+ torch.backends.cudnn.allow_tf32 = False
42
+ torch.backends.mha.set_fastpath_enabled(False)
43
+ native = load_native(args.native, expected_manifest_sha256=args.manifest_sha256, device='cuda:0')
44
+ client = SystemOne(native, model=args.model, batching=args.batching)
45
+ result = client.batch(request) if isinstance(request, list) else client.evaluate(request)
46
+ with Path(args.output).open('x', encoding='utf-8') as stream:
47
+ json.dump(result, stream, ensure_ascii=False, allow_nan=False, indent=2)
48
+ stream.write('\n')
49
+
50
+
51
+ if __name__ == '__main__':
52
+ main()
53
+