Add System One typed batch API and matching fine-tuning inputs
Browse files- FINETUNING.md +2 -0
- PACKAGE_MANIFEST.json +49 -10
- README.md +3 -1
- SYSTEM_ONE.md +129 -0
- SYSTEM_ONE_VALIDATION.json +66 -0
- USAGE.md +42 -1
- decision_finetune/system_one.py +34 -0
- decision_inference/__init__.py +2 -2
- decision_inference/_system_one.py +217 -0
- examples/system-one.json +29 -0
- systemone.py +53 -0
FINETUNING.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
| 1 |
# Fine-tune on your data
|
| 2 |
|
|
|
|
|
|
|
| 3 |
The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
|
| 4 |
|
| 5 |
Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
|
|
|
|
| 1 |
# Fine-tune on your data
|
| 2 |
|
| 3 |
+
Already using System One requests? [Convert the same state/questions/criteria into training rows](SYSTEM_ONE.md#fine-tune-with-the-same-inputs) with separate hard or soft targets and shared source-component IDs.
|
| 4 |
+
|
| 5 |
The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
|
| 6 |
|
| 7 |
Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
|
PACKAGE_MANIFEST.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
-
"bytes_excluding_this_manifest":
|
| 3 |
-
"file_count_excluding_this_manifest":
|
| 4 |
"files": {
|
| 5 |
".gitattributes": {
|
| 6 |
"bytes": 753,
|
|
@@ -23,8 +23,8 @@
|
|
| 23 |
"sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
|
| 24 |
},
|
| 25 |
"FINETUNING.md": {
|
| 26 |
-
"bytes":
|
| 27 |
-
"sha256": "
|
| 28 |
},
|
| 29 |
"LICENSE": {
|
| 30 |
"bytes": 11358,
|
|
@@ -75,13 +75,21 @@
|
|
| 75 |
"sha256": "7e275e73c18b756a20b886272750b58416ead2c345b12393e162050231aa8f64"
|
| 76 |
},
|
| 77 |
"README.md": {
|
| 78 |
-
"bytes":
|
| 79 |
-
"sha256": "
|
| 80 |
},
|
| 81 |
"RUNTIME_UPDATE.json": {
|
| 82 |
"bytes": 1087,
|
| 83 |
"sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
|
| 84 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 85 |
"TECHNICAL_VALIDATION.json": {
|
| 86 |
"bytes": 1547,
|
| 87 |
"sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
|
|
@@ -95,8 +103,8 @@
|
|
| 95 |
"sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
|
| 96 |
},
|
| 97 |
"USAGE.md": {
|
| 98 |
-
"bytes":
|
| 99 |
-
"sha256": "
|
| 100 |
},
|
| 101 |
"VALIDATION.md": {
|
| 102 |
"bytes": 2120,
|
|
@@ -182,9 +190,13 @@
|
|
| 182 |
"bytes": 5751,
|
| 183 |
"sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
|
| 184 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
| 185 |
"decision_inference/__init__.py": {
|
| 186 |
-
"bytes":
|
| 187 |
-
"sha256": "
|
| 188 |
},
|
| 189 |
"decision_inference/_auto.py": {
|
| 190 |
"bytes": 1325,
|
|
@@ -194,6 +206,10 @@
|
|
| 194 |
"bytes": 3026,
|
| 195 |
"sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
|
| 196 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
| 197 |
"decision_inference/profile.py": {
|
| 198 |
"bytes": 2071,
|
| 199 |
"sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
|
|
@@ -222,6 +238,10 @@
|
|
| 222 |
"bytes": 1242,
|
| 223 |
"sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
|
| 224 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
| 225 |
"infer.py": {
|
| 226 |
"bytes": 2373,
|
| 227 |
"sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
|
|
@@ -385,6 +405,10 @@
|
|
| 385 |
"source-metadata/VELA_LICENSE_SOURCE_README.md": {
|
| 386 |
"bytes": 2046,
|
| 387 |
"sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
|
|
|
|
|
|
|
|
|
|
|
|
|
| 388 |
}
|
| 389 |
},
|
| 390 |
"language_scope": "English typed-decisions specialist",
|
|
@@ -402,6 +426,7 @@
|
|
| 402 |
},
|
| 403 |
"parameter_tensors": 489,
|
| 404 |
"parameters": 571909635,
|
|
|
|
| 405 |
"publication": {
|
| 406 |
"download_access": "public_ungated",
|
| 407 |
"new_contribution_license": "Apache-2.0",
|
|
@@ -415,6 +440,20 @@
|
|
| 415 |
"schema": "decision.public-distribution.v1",
|
| 416 |
"status": "READY_FOR_ROOT_PUBLICATION",
|
| 417 |
"subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 418 |
"technical_validation": "TECHNICAL_VALIDATION.json",
|
| 419 |
"training_provenance": "TRAINING_PROVENANCE.json"
|
| 420 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"bytes_excluding_this_manifest": 2327409686,
|
| 3 |
+
"file_count_excluding_this_manifest": 102,
|
| 4 |
"files": {
|
| 5 |
".gitattributes": {
|
| 6 |
"bytes": 753,
|
|
|
|
| 23 |
"sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
|
| 24 |
},
|
| 25 |
"FINETUNING.md": {
|
| 26 |
+
"bytes": 3842,
|
| 27 |
+
"sha256": "2874eedec3b19462552a35e0dc99af29c456b3a2f8096cc61344b92e4cc6db41"
|
| 28 |
},
|
| 29 |
"LICENSE": {
|
| 30 |
"bytes": 11358,
|
|
|
|
| 75 |
"sha256": "7e275e73c18b756a20b886272750b58416ead2c345b12393e162050231aa8f64"
|
| 76 |
},
|
| 77 |
"README.md": {
|
| 78 |
+
"bytes": 4416,
|
| 79 |
+
"sha256": "dd76e4f9879defc462441f899974140e9a65a6361f26dff883fcedfc473381af"
|
| 80 |
},
|
| 81 |
"RUNTIME_UPDATE.json": {
|
| 82 |
"bytes": 1087,
|
| 83 |
"sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
|
| 84 |
},
|
| 85 |
+
"SYSTEM_ONE.md": {
|
| 86 |
+
"bytes": 5735,
|
| 87 |
+
"sha256": "414e12680a6ec2dbbbc660d1f6ba0d601fa941388476a62696b7f5e8a3e11ef9"
|
| 88 |
+
},
|
| 89 |
+
"SYSTEM_ONE_VALIDATION.json": {
|
| 90 |
+
"bytes": 2583,
|
| 91 |
+
"sha256": "24c28a2b8c448507f26e116a7f6a6d4e40fb8eaae0384e6bee421a105e415e0a"
|
| 92 |
+
},
|
| 93 |
"TECHNICAL_VALIDATION.json": {
|
| 94 |
"bytes": 1547,
|
| 95 |
"sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
|
|
|
|
| 103 |
"sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
|
| 104 |
},
|
| 105 |
"USAGE.md": {
|
| 106 |
+
"bytes": 4589,
|
| 107 |
+
"sha256": "581e65562cdf35d8eac544640b984e6267d86e3b46adcaf1a00bd9d9c5af247c"
|
| 108 |
},
|
| 109 |
"VALIDATION.md": {
|
| 110 |
"bytes": 2120,
|
|
|
|
| 190 |
"bytes": 5751,
|
| 191 |
"sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
|
| 192 |
},
|
| 193 |
+
"decision_finetune/system_one.py": {
|
| 194 |
+
"bytes": 1784,
|
| 195 |
+
"sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
|
| 196 |
+
},
|
| 197 |
"decision_inference/__init__.py": {
|
| 198 |
+
"bytes": 282,
|
| 199 |
+
"sha256": "8199829090f1a3ca5fe59aec1a6dce2beb57026c301d2c87d8c5b1c960981e66"
|
| 200 |
},
|
| 201 |
"decision_inference/_auto.py": {
|
| 202 |
"bytes": 1325,
|
|
|
|
| 206 |
"bytes": 3026,
|
| 207 |
"sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
|
| 208 |
},
|
| 209 |
+
"decision_inference/_system_one.py": {
|
| 210 |
+
"bytes": 10690,
|
| 211 |
+
"sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
|
| 212 |
+
},
|
| 213 |
"decision_inference/profile.py": {
|
| 214 |
"bytes": 2071,
|
| 215 |
"sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
|
|
|
|
| 238 |
"bytes": 1242,
|
| 239 |
"sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
|
| 240 |
},
|
| 241 |
+
"examples/system-one.json": {
|
| 242 |
+
"bytes": 692,
|
| 243 |
+
"sha256": "0b8320ecbeded8c3229dccc52f5045f00e8d1968183bc9d1371d24e949925dcf"
|
| 244 |
+
},
|
| 245 |
"infer.py": {
|
| 246 |
"bytes": 2373,
|
| 247 |
"sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
|
|
|
|
| 405 |
"source-metadata/VELA_LICENSE_SOURCE_README.md": {
|
| 406 |
"bytes": 2046,
|
| 407 |
"sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
|
| 408 |
+
},
|
| 409 |
+
"systemone.py": {
|
| 410 |
+
"bytes": 2368,
|
| 411 |
+
"sha256": "8ce311c5651f1b0c1bc99d5a4ccaa6c7878b5c509256e4668a2191ad8c35e2e3"
|
| 412 |
}
|
| 413 |
},
|
| 414 |
"language_scope": "English typed-decisions specialist",
|
|
|
|
| 426 |
},
|
| 427 |
"parameter_tensors": 489,
|
| 428 |
"parameters": 571909635,
|
| 429 |
+
"public_non_native_bytes_excluding_this_manifest": 5150475,
|
| 430 |
"publication": {
|
| 431 |
"download_access": "public_ungated",
|
| 432 |
"new_contribution_license": "Apache-2.0",
|
|
|
|
| 440 |
"schema": "decision.public-distribution.v1",
|
| 441 |
"status": "READY_FOR_ROOT_PUBLICATION",
|
| 442 |
"subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 443 |
+
"system_one_api": {
|
| 444 |
+
"default_physical_batch_limit": 8,
|
| 445 |
+
"entry": "decision_inference.SystemOne",
|
| 446 |
+
"finetune_conversion": "decision_finetune.system_one.system_one_training_rows",
|
| 447 |
+
"max_batch_decisions": 512,
|
| 448 |
+
"max_questions": 128,
|
| 449 |
+
"max_request_bytes": 2097152,
|
| 450 |
+
"max_requests": 128,
|
| 451 |
+
"optional_auto_capacity": 32,
|
| 452 |
+
"request": "state/model/questions",
|
| 453 |
+
"response": "model/answers/usage",
|
| 454 |
+
"validation": "SYSTEM_ONE_VALIDATION.json",
|
| 455 |
+
"weights_unchanged": true
|
| 456 |
+
},
|
| 457 |
"technical_validation": "TECHNICAL_VALIDATION.json",
|
| 458 |
"training_provenance": "TRAINING_PROVENANCE.json"
|
| 459 |
}
|
README.md
CHANGED
|
@@ -54,9 +54,11 @@ Lex is evaluated as an English specialist on these four workflows. For broader m
|
|
| 54 |
|
| 55 |
## Make it yours
|
| 56 |
|
|
|
|
|
|
|
| 57 |
Run the [Python examples](USAGE.md), or adapt Lex to your own labels and rubrics with the included [fine-tuning CLI](FINETUNING.md). Both hard and soft training labels are supported, with checkpoint resume.
|
| 58 |
|
| 59 |
-
The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the Studio
|
| 60 |
|
| 61 |
## Architecture
|
| 62 |
|
|
|
|
| 54 |
|
| 55 |
## Make it yours
|
| 56 |
|
| 57 |
+
**One state. Many decisions.** Use the [System One API](SYSTEM_ONE.md) to submit up to 128 typed questions, or batch the same questions across independent contexts. Results return under your original question IDs.
|
| 58 |
+
|
| 59 |
Run the [Python examples](USAGE.md), or adapt Lex to your own labels and rubrics with the included [fine-tuning CLI](FINETUNING.md). Both hard and soft training labels are supported, with checkpoint resume.
|
| 60 |
|
| 61 |
+
The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
|
| 62 |
|
| 63 |
## Architecture
|
| 64 |
|
SYSTEM_ONE.md
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# One state. Many decisions.
|
| 2 |
+
|
| 3 |
+
Decision uses the [System One API format](https://docs.typesafe.ai/api): supply
|
| 4 |
+
`state`, `model` and a map of typed `questions`; receive an `answers` map under
|
| 5 |
+
the same question IDs. Noul returns the probability of yes, Choice selects from
|
| 6 |
+
your named options, and Score evaluates an ordered rubric.
|
| 7 |
+
|
| 8 |
+
```python
|
| 9 |
+
from decision_inference import SystemOne
|
| 10 |
+
|
| 11 |
+
# native is your loaded Decision checkpoint; see USAGE.md for loading.
|
| 12 |
+
client = SystemOne(native, batching="auto")
|
| 13 |
+
result = client.system_one(
|
| 14 |
+
state={"message": "Please refund the duplicate charge. I need this fixed today."},
|
| 15 |
+
questions={
|
| 16 |
+
"refund_requested": {
|
| 17 |
+
"type": "noul",
|
| 18 |
+
"instructions": "Does the customer explicitly request a refund?",
|
| 19 |
+
},
|
| 20 |
+
"team": {
|
| 21 |
+
"type": "choice",
|
| 22 |
+
"instructions": "Which team should handle this request?",
|
| 23 |
+
"criteria": {"Billing": "Charges and refunds", "Support": "Technical problems"},
|
| 24 |
+
},
|
| 25 |
+
"urgency": {
|
| 26 |
+
"type": "score",
|
| 27 |
+
"instructions": "How urgent is the request?",
|
| 28 |
+
"criteria": ["No deadline", "Needed soon", "Needed today"],
|
| 29 |
+
},
|
| 30 |
+
},
|
| 31 |
+
)
|
| 32 |
+
print(result["answers"])
|
| 33 |
+
```
|
| 34 |
+
|
| 35 |
+
`client.evaluate(request)` accepts the complete JSON request object, including
|
| 36 |
+
`model`. A client is bound to its loaded checkpoint; a mismatched model is an
|
| 37 |
+
error. Public Kai/Lex names are inferred from their verified native manifest.
|
| 38 |
+
For your own fine-tune, use `SystemOne(native, model="my-decision-model")`.
|
| 39 |
+
|
| 40 |
+
## Batch the questions or the contexts
|
| 41 |
+
|
| 42 |
+
One request accepts **up to 128 questions**. For one set of questions across many
|
| 43 |
+
independent contexts, use `client.batch`:
|
| 44 |
+
|
| 45 |
+
```python
|
| 46 |
+
questions = {
|
| 47 |
+
"refund_requested": {
|
| 48 |
+
"type": "noul",
|
| 49 |
+
"instructions": "Does the customer explicitly request a refund?",
|
| 50 |
+
}
|
| 51 |
+
}
|
| 52 |
+
results = client.batch([
|
| 53 |
+
{"model": client.model, "state": message, "questions": questions}
|
| 54 |
+
for message in messages
|
| 55 |
+
])
|
| 56 |
+
# results[i] corresponds to messages[i]; question IDs can repeat across requests.
|
| 57 |
+
```
|
| 58 |
+
|
| 59 |
+
`batch` is a Decision Python extension around ordinary System One requests. It
|
| 60 |
+
accepts up to 128 requests and 512 total decisions, with a combined 2 MiB input
|
| 61 |
+
limit. Responses preserve request order and question order. The entire batch
|
| 62 |
+
must pass validation and complete-input token admission before any forward.
|
| 63 |
+
An overlength or malformed input fails the call without partial answers.
|
| 64 |
+
|
| 65 |
+
The default uses physical batches of up to eight. `batching="auto"` uses the
|
| 66 |
+
released padding-aware scheduler: up to 32 consecutive same-type decisions when
|
| 67 |
+
that adds no padding; otherwise it keeps B8. FP32 rounding can vary with physical
|
| 68 |
+
batch shape. Each state/question pair still has its own encoder computation.
|
| 69 |
+
One API call does not imply one forward or a shared state activation cache.
|
| 70 |
+
|
| 71 |
+
## Typed fields
|
| 72 |
+
|
| 73 |
+
| Type | `criteria` | Answer |
|
| 74 |
+
|---|---|---|
|
| 75 |
+
| `noul` | Optional `true` / `false` descriptions | `type`, `noul` |
|
| 76 |
+
| `choice` | 2–255 named options; descriptions may be null | `type`, `choice`, `probabilities`, `confidence` |
|
| 77 |
+
| `score` | 2–10 ordered level descriptions | `type`, `score`, `legend`, `probabilities`, `confidence` |
|
| 78 |
+
|
| 79 |
+
State, instructions and descriptions accept strings, JSON objects or arrays.
|
| 80 |
+
Structured values become deterministic compact JSON with sorted object keys;
|
| 81 |
+
array order, Choice option order and Score level order are preserved. Question
|
| 82 |
+
IDs are bookkeeping only. Choice names are part of the semantic input, including
|
| 83 |
+
when their description is null. Score levels are indexed from zero; `score`
|
| 84 |
+
preserves the native probability-weighted FP32 expectation. Structured Score
|
| 85 |
+
descriptions appear as JSON strings in `legend`.
|
| 86 |
+
|
| 87 |
+
Every complete state/question/candidate sequence must fit **1,024 tokens**.
|
| 88 |
+
Nothing is truncated. The token count includes the repeated state for each
|
| 89 |
+
question; `usage.input_tokens` sums these complete sequences and
|
| 90 |
+
`usage.output_tokens` is zero because the model returns scores without generating
|
| 91 |
+
text. These are local computation counts, not TypeSafe billing counts.
|
| 92 |
+
|
| 93 |
+
Decision's `confidence` is the largest candidate probability, matching its native
|
| 94 |
+
runtime. It is not calibrated correctness. TypeSafe does not specify its own
|
| 95 |
+
formula in the [confidence documentation](https://docs.typesafe.ai/confidence),
|
| 96 |
+
so thresholds should not be transferred between models without validation.
|
| 97 |
+
The shared request/answer schema does not claim identical weights, confidence
|
| 98 |
+
values, hosted service limits or SDK behavior.
|
| 99 |
+
|
| 100 |
+
## Fine-tune with the same inputs
|
| 101 |
+
|
| 102 |
+
Use the same conversion for training so structured inputs, option names and
|
| 103 |
+
criteria have identical semantics at training and inference time:
|
| 104 |
+
|
| 105 |
+
```python
|
| 106 |
+
import json
|
| 107 |
+
from pathlib import Path
|
| 108 |
+
from decision_finetune.system_one import system_one_training_rows
|
| 109 |
+
|
| 110 |
+
request = json.loads(Path("examples/system-one.json").read_text())
|
| 111 |
+
rows = system_one_training_rows(
|
| 112 |
+
request, # the same model/state/questions object used for inference
|
| 113 |
+
targets={
|
| 114 |
+
"refund_requested": {"probability": 1.0},
|
| 115 |
+
"team": {"choice_id": "Billing"},
|
| 116 |
+
"urgency": {"probabilities": [0.0, 0.0, 1.0]},
|
| 117 |
+
},
|
| 118 |
+
request_id="ticket-1001",
|
| 119 |
+
source_id="support-tickets",
|
| 120 |
+
component_id="customer-42",
|
| 121 |
+
)
|
| 122 |
+
```
|
| 123 |
+
|
| 124 |
+
Save the rows as JSONL for the existing [fine-tuning CLI](FINETUNING.md). Targets
|
| 125 |
+
stay separate from state, instructions and criteria. Related examples share a
|
| 126 |
+
source/component ID and stay in one split; the CLI checks component and exact
|
| 127 |
+
input overlap. Optional `hard_target_ids` supplies separate evaluation labels
|
| 128 |
+
under the same question IDs. Noul supports soft yes probabilities; Choice and
|
| 129 |
+
Score support complete soft distributions in the original candidate order.
|
SYSTEM_ONE_VALIDATION.json
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"all_reference_outputs_exact": true,
|
| 3 |
+
"api_source": {
|
| 4 |
+
"bytes": 10690,
|
| 5 |
+
"sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
|
| 6 |
+
},
|
| 7 |
+
"complete_input_tokens": 1024,
|
| 8 |
+
"coverage": [
|
| 9 |
+
"Noul, Choice and Score",
|
| 10 |
+
"128 questions",
|
| 11 |
+
"128 contexts and 512 decisions",
|
| 12 |
+
"question rename and reorder",
|
| 13 |
+
"irrelevant question insertion",
|
| 14 |
+
"single and multi-context agreement",
|
| 15 |
+
"default B8 and opt-in padding-aware B32",
|
| 16 |
+
"late invalid input before any forward",
|
| 17 |
+
"same training and inference input encoding"
|
| 18 |
+
],
|
| 19 |
+
"cross_batch_tolerance": 2e-05,
|
| 20 |
+
"device": "AMD GPU / ROCm",
|
| 21 |
+
"limits": "Technical compatibility validation on synthetic unlabeled cases; not evidence of accuracy, calibration or latency gains.",
|
| 22 |
+
"max_decisions_per_batch": 512,
|
| 23 |
+
"max_questions_per_request": 128,
|
| 24 |
+
"max_requests_per_batch": 128,
|
| 25 |
+
"models": {
|
| 26 |
+
"Kai": {
|
| 27 |
+
"all_489_parameters_and_66_buffers_unchanged": true,
|
| 28 |
+
"cross_case_checks": 713,
|
| 29 |
+
"forward_calls": 242,
|
| 30 |
+
"forward_rows": 2756,
|
| 31 |
+
"hard_flips": 0,
|
| 32 |
+
"invalid_input_checks": 7,
|
| 33 |
+
"invalid_input_forward_calls": 0,
|
| 34 |
+
"max_probability_error": 2.0563602447509766e-06,
|
| 35 |
+
"max_score_error": 2.980232238769531e-07,
|
| 36 |
+
"native_manifest_sha256": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
|
| 37 |
+
"result_sha256": "8cb3bc435e36b79d96b60058e9b4c57b9996dcb8a55a1cd3faf6c7324e7f917c",
|
| 38 |
+
"same_physical_reference_outputs_exact": true
|
| 39 |
+
},
|
| 40 |
+
"Lex": {
|
| 41 |
+
"all_489_parameters_and_66_buffers_unchanged": true,
|
| 42 |
+
"cross_case_checks": 713,
|
| 43 |
+
"forward_calls": 242,
|
| 44 |
+
"forward_rows": 2756,
|
| 45 |
+
"hard_flips": 0,
|
| 46 |
+
"invalid_input_checks": 7,
|
| 47 |
+
"invalid_input_forward_calls": 0,
|
| 48 |
+
"max_probability_error": 1.1324882507324219e-06,
|
| 49 |
+
"max_score_error": 7.152557373046875e-07,
|
| 50 |
+
"native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 51 |
+
"result_sha256": "e94c5f1109f15005415a48623602d5255b38bd33451b55c758a0db0c16f68191",
|
| 52 |
+
"same_physical_reference_outputs_exact": true
|
| 53 |
+
}
|
| 54 |
+
},
|
| 55 |
+
"optimizer_updates": 0,
|
| 56 |
+
"performance_benchmark": false,
|
| 57 |
+
"quality_evaluation": false,
|
| 58 |
+
"reference": "Independent System One mapping using the original Studio single-question conversion and native predictor; identical physical batches.",
|
| 59 |
+
"status": "PASS_AMD_SYSTEM_ONE_API",
|
| 60 |
+
"total_forward_calls": 484,
|
| 61 |
+
"training_conversion_source": {
|
| 62 |
+
"bytes": 1784,
|
| 63 |
+
"sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
|
| 64 |
+
},
|
| 65 |
+
"weights_unchanged": true
|
| 66 |
+
}
|
USAGE.md
CHANGED
|
@@ -7,12 +7,53 @@ hf download llm-semantic-router/Decision-1.0-Lex --local-dir Decision-1.0-Lex
|
|
| 7 |
cd Decision-1.0-Lex
|
| 8 |
```
|
| 9 |
|
| 10 |
-
Use `--local-dir` to materialize ordinary files for the native loader.
|
| 11 |
|
| 12 |
This initial release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
|
| 13 |
|
| 14 |
Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
|
| 15 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
```bash
|
| 17 |
export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
|
| 18 |
export PYTHONDONTWRITEBYTECODE=1
|
|
|
|
| 7 |
cd Decision-1.0-Lex
|
| 8 |
```
|
| 9 |
|
| 10 |
+
Use `--local-dir` to materialize ordinary files for the native loader. Use the compatible ROCm/Python environment described below.
|
| 11 |
|
| 12 |
This initial release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
|
| 13 |
|
| 14 |
Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
|
| 15 |
|
| 16 |
+
## System One: parallel typed questions
|
| 17 |
+
|
| 18 |
+
Use the standard `state / model / questions` request shape. The response contains
|
| 19 |
+
`model`, an `answers` map under the same question IDs, and `usage`.
|
| 20 |
+
|
| 21 |
+
```bash
|
| 22 |
+
ROCR_VISIBLE_DEVICES=0 python systemone.py \
|
| 23 |
+
--input examples/system-one.json --output answers.json --batching auto
|
| 24 |
+
```
|
| 25 |
+
|
| 26 |
+
```python
|
| 27 |
+
from decision_runtime import load_native
|
| 28 |
+
from decision_inference import SystemOne
|
| 29 |
+
|
| 30 |
+
native = load_native(
|
| 31 |
+
"native",
|
| 32 |
+
expected_manifest_sha256="f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 33 |
+
device="cuda:0",
|
| 34 |
+
)
|
| 35 |
+
client = SystemOne(native, batching="auto")
|
| 36 |
+
answers = client.system_one(
|
| 37 |
+
state="Please refund the duplicate charge. I need this fixed today.",
|
| 38 |
+
questions={
|
| 39 |
+
"refund": {"type": "noul", "instructions": "Is a refund requested?"},
|
| 40 |
+
"urgent": {"type": "noul", "instructions": "Does the customer specify a deadline?"},
|
| 41 |
+
},
|
| 42 |
+
)["answers"]
|
| 43 |
+
```
|
| 44 |
+
|
| 45 |
+
Up to 128 questions per request. `client.batch(requests)` combines independent
|
| 46 |
+
contexts with up to 512 total decisions. Each complete state/question/candidate
|
| 47 |
+
sequence must fit 1,024 tokens; validation happens before the first forward.
|
| 48 |
+
Default physical batching is B8; `auto` opts into the released homogeneous,
|
| 49 |
+
padding-aware B32 scheduler. [Fields, batching and the matching fine-tuning
|
| 50 |
+
conversion](SYSTEM_ONE.md).
|
| 51 |
+
|
| 52 |
+
## Native records
|
| 53 |
+
|
| 54 |
+
The original native JSONL interface remains available for applications that
|
| 55 |
+
need logits, explicit candidate IDs or arbitrary ordered Score values.
|
| 56 |
+
|
| 57 |
```bash
|
| 58 |
export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
|
| 59 |
export PYTHONDONTWRITEBYTECODE=1
|
decision_finetune/system_one.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Keep System One inference and fine-tuning input semantics identical."""
|
| 2 |
+
import copy
|
| 3 |
+
import json
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
def system_one_training_rows(request, targets, *, request_id, source_id,
|
| 7 |
+
component_id, hard_target_ids=None):
|
| 8 |
+
"""Convert a typed request plus separate native targets to training rows.
|
| 9 |
+
|
| 10 |
+
targets is keyed by question ID. Noul uses {'probability': p_yes}, Choice
|
| 11 |
+
{'choice_id': label} or {'probabilities': [...]}, Score {'probabilities': [...]}.
|
| 12 |
+
The caller assigns shared source/component IDs across related examples so
|
| 13 |
+
existing TRAIN/DEV validation can reject component overlap.
|
| 14 |
+
"""
|
| 15 |
+
from decision_inference._system_one import system_one_records
|
| 16 |
+
from .data import semantics
|
| 17 |
+
for value in (request_id, source_id, component_id):
|
| 18 |
+
if not isinstance(value, str) or not value.strip():
|
| 19 |
+
raise ValueError("Explicit nonempty request/source/component IDs required")
|
| 20 |
+
rows = system_one_records(request)
|
| 21 |
+
qids = {row["question"]["id"] for row in rows}
|
| 22 |
+
if not isinstance(targets, dict) or set(targets) != qids:
|
| 23 |
+
raise ValueError("Exactly one target per question ID is required")
|
| 24 |
+
if hard_target_ids is not None and (not isinstance(hard_target_ids, dict) or set(hard_target_ids) != qids):
|
| 25 |
+
raise ValueError("Hard evaluation labels must cover exactly the question IDs")
|
| 26 |
+
for i, row in enumerate(rows):
|
| 27 |
+
qid = row["question"]["id"]
|
| 28 |
+
row.update(id=json.dumps([request_id, i], ensure_ascii=False, separators=(",", ":")),
|
| 29 |
+
source_id=source_id, component_id=component_id,
|
| 30 |
+
target=copy.deepcopy(targets[qid]))
|
| 31 |
+
if hard_target_ids is not None:
|
| 32 |
+
row["hard_target_id"] = hard_target_ids[qid]
|
| 33 |
+
semantics(row)
|
| 34 |
+
return rows
|
decision_inference/__init__.py
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
|
| 2 |
-
|
| 3 |
from ._auto import predict_auto_1k
|
|
|
|
| 4 |
|
| 5 |
-
__all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k"]
|
|
|
|
| 1 |
from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
|
|
|
|
| 2 |
from ._auto import predict_auto_1k
|
| 3 |
+
from ._system_one import SystemOne, system_one_records
|
| 4 |
|
| 5 |
+
__all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k", "SystemOne", "system_one_records"]
|
decision_inference/_system_one.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""System One request/answer schema over complete-input Decision inference.
|
| 2 |
+
|
| 3 |
+
Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping;
|
| 4 |
+
Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache.
|
| 5 |
+
"""
|
| 6 |
+
import copy
|
| 7 |
+
import json
|
| 8 |
+
import math
|
| 9 |
+
|
| 10 |
+
MAX_QUESTIONS = 128
|
| 11 |
+
MAX_REQUESTS = 128
|
| 12 |
+
MAX_DECISIONS = 512
|
| 13 |
+
MAX_REQUEST_BYTES = 2 * 1024 * 1024
|
| 14 |
+
PUBLIC_MODELS = {
|
| 15 |
+
"Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
|
| 16 |
+
"Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 17 |
+
}
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _identifier(value, label):
|
| 21 |
+
if not isinstance(value, str) or not value.strip() or len(value) > 128:
|
| 22 |
+
raise ValueError(label + " must be a nonempty string of at most 128 characters")
|
| 23 |
+
return value
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def _json(value):
|
| 27 |
+
# JSON objects must have string keys: never silently coerce Python keys.
|
| 28 |
+
def check(item):
|
| 29 |
+
if isinstance(item, dict):
|
| 30 |
+
if not all(isinstance(k, str) for k in item):
|
| 31 |
+
raise ValueError("JSON object keys must be strings")
|
| 32 |
+
for v in item.values():
|
| 33 |
+
check(v)
|
| 34 |
+
elif isinstance(item, list):
|
| 35 |
+
for v in item:
|
| 36 |
+
check(v)
|
| 37 |
+
elif item is not None and not isinstance(item, (str, bool, int, float)):
|
| 38 |
+
raise ValueError("Only JSON values are supported")
|
| 39 |
+
try:
|
| 40 |
+
check(value)
|
| 41 |
+
return json.dumps(value, ensure_ascii=False, sort_keys=True,
|
| 42 |
+
separators=(",", ":"), allow_nan=False)
|
| 43 |
+
except (TypeError, RecursionError, UnicodeError) as exc:
|
| 44 |
+
raise ValueError("Invalid JSON content") from exc
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
def _content(value, label):
|
| 48 |
+
if isinstance(value, str):
|
| 49 |
+
if not value.strip():
|
| 50 |
+
raise ValueError(label + " must not be empty")
|
| 51 |
+
return value
|
| 52 |
+
if isinstance(value, (dict, list)):
|
| 53 |
+
return _json(value)
|
| 54 |
+
raise ValueError(label + " must be text, an object, or an array")
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def system_one_records(request):
|
| 58 |
+
"""Validate one wire request and return native rows, without loading a model.
|
| 59 |
+
|
| 60 |
+
Full token admission occurs in predict_1k before the first model forward.
|
| 61 |
+
External record IDs should be made unique when combining training examples.
|
| 62 |
+
"""
|
| 63 |
+
if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}:
|
| 64 |
+
raise ValueError("A request contains exactly model, state and questions")
|
| 65 |
+
_identifier(request["model"], "Model")
|
| 66 |
+
if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES:
|
| 67 |
+
raise ValueError("Request exceeds 2 MiB; no input is truncated")
|
| 68 |
+
state = _content(request["state"], "State")
|
| 69 |
+
questions = request["questions"]
|
| 70 |
+
if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS:
|
| 71 |
+
raise ValueError("Provide 1..128 named questions")
|
| 72 |
+
rows = []
|
| 73 |
+
for index, (qid, item) in enumerate(questions.items()):
|
| 74 |
+
_identifier(qid, "Question ID")
|
| 75 |
+
if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"}
|
| 76 |
+
or not {"type", "instructions"} <= set(item)):
|
| 77 |
+
raise ValueError(qid + ": use type, instructions and optional criteria")
|
| 78 |
+
kind = item["type"]
|
| 79 |
+
if kind not in ("noul", "choice", "score"):
|
| 80 |
+
raise ValueError(qid + ": type must be noul, choice or score")
|
| 81 |
+
q = {"id": qid, "type": kind.capitalize(),
|
| 82 |
+
"text": _content(item["instructions"], qid + ".instructions")}
|
| 83 |
+
criteria = item.get("criteria")
|
| 84 |
+
if kind == "choice":
|
| 85 |
+
if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255:
|
| 86 |
+
raise ValueError(qid + ": Choice requires 2..255 named options")
|
| 87 |
+
q["options"] = []
|
| 88 |
+
for name, description in criteria.items():
|
| 89 |
+
_identifier(name, "Choice option")
|
| 90 |
+
text = name if description is None else name + ": " + _content(description, qid + ".criteria")
|
| 91 |
+
q["options"].append({"id": name, "text": text})
|
| 92 |
+
elif kind == "score":
|
| 93 |
+
if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
|
| 94 |
+
raise ValueError(qid + ": Score requires 2..10 ordered levels")
|
| 95 |
+
q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")}
|
| 96 |
+
for i, v in enumerate(criteria)]
|
| 97 |
+
elif "criteria" in item:
|
| 98 |
+
if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
|
| 99 |
+
raise ValueError(qid + ": Noul criteria accept false and true only")
|
| 100 |
+
for key in ("false", "true"):
|
| 101 |
+
if key in criteria:
|
| 102 |
+
q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key)
|
| 103 |
+
rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q})
|
| 104 |
+
return rows
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def _answer(row, prediction):
|
| 108 |
+
q = row["question"]
|
| 109 |
+
kind = q["type"].lower()
|
| 110 |
+
ids = (["no", "yes"] if kind == "noul" else
|
| 111 |
+
[v["id"] for v in q["options" if kind == "choice" else "levels"]])
|
| 112 |
+
if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"]
|
| 113 |
+
or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids
|
| 114 |
+
or type(prediction.get("input_tokens")) is not int
|
| 115 |
+
or not 1 <= prediction["input_tokens"] <= 1024
|
| 116 |
+
or type(prediction.get("state_tokens_original")) is not int
|
| 117 |
+
or prediction["state_tokens_original"] < 0
|
| 118 |
+
or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")):
|
| 119 |
+
raise RuntimeError("Prediction identity or complete-input profile mismatch")
|
| 120 |
+
p = prediction.get("probabilities")
|
| 121 |
+
if (not isinstance(p, list) or len(p) != len(ids)
|
| 122 |
+
or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p)
|
| 123 |
+
or abs(sum(p) - 1) > 2e-5):
|
| 124 |
+
raise RuntimeError("Invalid prediction probabilities")
|
| 125 |
+
answer = {"type": kind}
|
| 126 |
+
if kind == "noul":
|
| 127 |
+
if prediction.get("probability") != p[1]:
|
| 128 |
+
raise RuntimeError("Native Noul probability mismatch")
|
| 129 |
+
answer["noul"] = p[1]
|
| 130 |
+
return answer
|
| 131 |
+
best = ids[max(range(len(p)), key=p.__getitem__)]
|
| 132 |
+
if prediction.get("choice_id") != best or prediction.get("confidence") != max(p):
|
| 133 |
+
raise RuntimeError("Native Choice/confidence mismatch")
|
| 134 |
+
answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"])
|
| 135 |
+
if kind == "choice":
|
| 136 |
+
answer["choice"] = best
|
| 137 |
+
else:
|
| 138 |
+
score = prediction.get("score")
|
| 139 |
+
if (type(score) not in (int, float) or not math.isfinite(score)
|
| 140 |
+
or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5):
|
| 141 |
+
raise RuntimeError("Native ordinal Score mismatch")
|
| 142 |
+
# Preserve native FP32 arithmetic, not a new CPU reduction.
|
| 143 |
+
answer["score"] = score
|
| 144 |
+
answer["legend"] = {v["id"]: v["text"] for v in q["levels"]}
|
| 145 |
+
return answer
|
| 146 |
+
|
| 147 |
+
|
| 148 |
+
class SystemOne:
|
| 149 |
+
"""Local System One API for a loaded Kai, Lex or compatible fine-tune.
|
| 150 |
+
|
| 151 |
+
evaluate(request) accepts the HTTP body shape; system_one(**request) is its
|
| 152 |
+
Python equivalent. batch(requests) flattens independent states into GPU
|
| 153 |
+
batches and restores the original request/question order. B8 is the default;
|
| 154 |
+
batching='auto' opts into the published homogeneous padding-aware B32 path.
|
| 155 |
+
"""
|
| 156 |
+
def __init__(self, native, *, model=None, batching="default"):
|
| 157 |
+
if model is None:
|
| 158 |
+
model = next((name for name, sha in PUBLIC_MODELS.items()
|
| 159 |
+
if sha == native.manifest_sha256), None)
|
| 160 |
+
_identifier(model, "Model (required for a custom fine-tune)")
|
| 161 |
+
# Do not let a different loaded checkpoint claim a published identity.
|
| 162 |
+
if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]:
|
| 163 |
+
raise ValueError("Loaded checkpoint does not match the public model name")
|
| 164 |
+
if batching not in ("default", "auto"):
|
| 165 |
+
raise ValueError("batching must be default or auto")
|
| 166 |
+
self.native, self.model, self.batching = native, model, batching
|
| 167 |
+
|
| 168 |
+
def system_one(self, *, state, questions, model=None):
|
| 169 |
+
return self.evaluate({"model": self.model if model is None else model,
|
| 170 |
+
"state": state, "questions": questions})
|
| 171 |
+
|
| 172 |
+
def evaluate(self, request):
|
| 173 |
+
return self.batch([request])[0]
|
| 174 |
+
|
| 175 |
+
def batch(self, requests):
|
| 176 |
+
if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS:
|
| 177 |
+
raise ValueError("Provide 1..128 request objects")
|
| 178 |
+
if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES:
|
| 179 |
+
raise ValueError("Combined request exceeds 2 MiB")
|
| 180 |
+
# Detach mutable caller inputs before conversion/admission/inference.
|
| 181 |
+
requests = copy.deepcopy(requests)
|
| 182 |
+
groups = []
|
| 183 |
+
for request in requests:
|
| 184 |
+
rows = system_one_records(request)
|
| 185 |
+
if request["model"] != self.model:
|
| 186 |
+
raise ValueError("Request model does not match this loaded model")
|
| 187 |
+
groups.append(rows)
|
| 188 |
+
count = sum(map(len, groups))
|
| 189 |
+
if count > MAX_DECISIONS:
|
| 190 |
+
raise ValueError("Provide at most 512 decisions in one batch")
|
| 191 |
+
records, slots = [], []
|
| 192 |
+
# Question-major order permits the same question across many states to
|
| 193 |
+
# share a physical batch; external IDs never decide caching or grouping.
|
| 194 |
+
for qi in range(max(map(len, groups))):
|
| 195 |
+
for ri, group in enumerate(groups):
|
| 196 |
+
if qi < len(group):
|
| 197 |
+
row = group[qi]
|
| 198 |
+
row["id"] = f"systemone:{ri}:{qi}"
|
| 199 |
+
records.append(row)
|
| 200 |
+
slots.append((ri, qi))
|
| 201 |
+
if self.batching == "auto":
|
| 202 |
+
from ._auto import predict_auto_1k
|
| 203 |
+
predictions = predict_auto_1k(self.native, records)
|
| 204 |
+
else:
|
| 205 |
+
from ._request import predict_1k
|
| 206 |
+
predictions = predict_1k(self.native, records, batch_size=8)
|
| 207 |
+
if len(predictions) != len(records):
|
| 208 |
+
raise RuntimeError("Incomplete model result; no partial answers returned")
|
| 209 |
+
values = [[None] * len(group) for group in groups]
|
| 210 |
+
tokens = [0] * len(groups)
|
| 211 |
+
for row, prediction, (ri, qi) in zip(records, predictions, slots):
|
| 212 |
+
values[ri][qi] = _answer(row, prediction)
|
| 213 |
+
tokens[ri] += prediction["input_tokens"]
|
| 214 |
+
return [{"model": self.model,
|
| 215 |
+
"answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])},
|
| 216 |
+
"usage": {"input_tokens": tokens[ri], "output_tokens": 0}}
|
| 217 |
+
for ri, group in enumerate(groups)]
|
examples/system-one.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "Decision-1.0-Lex",
|
| 3 |
+
"state": {
|
| 4 |
+
"message": "Please refund the duplicate charge. I need this fixed today."
|
| 5 |
+
},
|
| 6 |
+
"questions": {
|
| 7 |
+
"refund_requested": {
|
| 8 |
+
"type": "noul",
|
| 9 |
+
"instructions": "Does the customer explicitly request a refund?"
|
| 10 |
+
},
|
| 11 |
+
"team": {
|
| 12 |
+
"type": "choice",
|
| 13 |
+
"instructions": "Which team should handle this request?",
|
| 14 |
+
"criteria": {
|
| 15 |
+
"Billing": "Charges and refunds",
|
| 16 |
+
"Support": "Technical problems"
|
| 17 |
+
}
|
| 18 |
+
},
|
| 19 |
+
"urgency": {
|
| 20 |
+
"type": "score",
|
| 21 |
+
"instructions": "How urgent is the request?",
|
| 22 |
+
"criteria": [
|
| 23 |
+
"No deadline",
|
| 24 |
+
"Needed soon",
|
| 25 |
+
"Needed today"
|
| 26 |
+
]
|
| 27 |
+
}
|
| 28 |
+
}
|
| 29 |
+
}
|
systemone.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Run one System One request, or a batch of requests, on a local Decision model."""
|
| 2 |
+
import argparse
|
| 3 |
+
import json
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
NATIVE_MANIFEST_SHA256 = 'f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6'
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def _unique_object(pairs):
|
| 10 |
+
result = {}
|
| 11 |
+
for key, value in pairs:
|
| 12 |
+
if key in result:
|
| 13 |
+
raise ValueError('Duplicate JSON object key: ' + key)
|
| 14 |
+
result[key] = value
|
| 15 |
+
return result
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
def main():
|
| 19 |
+
parser = argparse.ArgumentParser(description='Decision System One: typed parallel questions on AMD')
|
| 20 |
+
parser.add_argument('--native', default=str(Path(__file__).resolve().parent / 'native'))
|
| 21 |
+
parser.add_argument('--manifest-sha256', default=NATIVE_MANIFEST_SHA256)
|
| 22 |
+
parser.add_argument('--model', help='Explicit name for a compatible custom fine-tune')
|
| 23 |
+
parser.add_argument('--input', required=True, help='JSON request or array of request objects')
|
| 24 |
+
parser.add_argument('--output', required=True, help='Fresh response JSON file')
|
| 25 |
+
parser.add_argument('--batching', choices=('default', 'auto'), default='default')
|
| 26 |
+
args = parser.parse_args()
|
| 27 |
+
payload = Path(args.input).read_bytes()
|
| 28 |
+
if len(payload) > 2 * 1024 * 1024:
|
| 29 |
+
raise ValueError('Input file exceeds 2 MiB')
|
| 30 |
+
request = json.loads(payload, object_pairs_hook=_unique_object)
|
| 31 |
+
if Path(args.output).exists():
|
| 32 |
+
raise ValueError('Output must be fresh')
|
| 33 |
+
import torch
|
| 34 |
+
from decision_runtime import load_native
|
| 35 |
+
from decision_inference import SystemOne
|
| 36 |
+
if torch.version.hip is None or not torch.cuda.is_available() or torch.cuda.device_count() != 1:
|
| 37 |
+
raise RuntimeError('Expose exactly one AMD ROCm GPU')
|
| 38 |
+
torch.cuda.set_device(0)
|
| 39 |
+
torch.set_num_threads(2)
|
| 40 |
+
torch.backends.cuda.matmul.allow_tf32 = False
|
| 41 |
+
torch.backends.cudnn.allow_tf32 = False
|
| 42 |
+
torch.backends.mha.set_fastpath_enabled(False)
|
| 43 |
+
native = load_native(args.native, expected_manifest_sha256=args.manifest_sha256, device='cuda:0')
|
| 44 |
+
client = SystemOne(native, model=args.model, batching=args.batching)
|
| 45 |
+
result = client.batch(request) if isinstance(request, list) else client.evaluate(request)
|
| 46 |
+
with Path(args.output).open('x', encoding='utf-8') as stream:
|
| 47 |
+
json.dump(result, stream, ensure_ascii=False, allow_nan=False, indent=2)
|
| 48 |
+
stream.write('\n')
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
if __name__ == '__main__':
|
| 52 |
+
main()
|
| 53 |
+
|