File size: 10,728 Bytes
a2af77f ee8e74d a2af77f ee8e74d a2af77f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 | """System One request/answer schema over complete-input Decision inference.
Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping;
Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache.
"""
import copy
import json
import math
MAX_QUESTIONS = 128
MAX_REQUESTS = 128
MAX_DECISIONS = 512
MAX_REQUEST_BYTES = 2 * 1024 * 1024
PUBLIC_MODELS = {
"Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
"Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
}
def _identifier(value, label):
if not isinstance(value, str) or not value.strip() or len(value) > 128:
raise ValueError(label + " must be a nonempty string of at most 128 characters")
return value
def _json(value):
# JSON objects must have string keys: never silently coerce Python keys.
def check(item):
if isinstance(item, dict):
if not all(isinstance(k, str) for k in item):
raise ValueError("JSON object keys must be strings")
for v in item.values():
check(v)
elif isinstance(item, list):
for v in item:
check(v)
elif item is not None and not isinstance(item, (str, bool, int, float)):
raise ValueError("Only JSON values are supported")
try:
check(value)
return json.dumps(value, ensure_ascii=False, sort_keys=True,
separators=(",", ":"), allow_nan=False)
except (TypeError, RecursionError, UnicodeError) as exc:
raise ValueError("Invalid JSON content") from exc
def _content(value, label):
if isinstance(value, str):
if not value.strip():
raise ValueError(label + " must not be empty")
return value
if isinstance(value, (dict, list)):
return _json(value)
raise ValueError(label + " must be text, an object, or an array")
def system_one_records(request):
"""Validate one wire request and return native rows, without loading a model.
Full token admission occurs in predict_1k before the first model forward.
External record IDs should be made unique when combining training examples.
"""
if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}:
raise ValueError("A request contains exactly model, state and questions")
_identifier(request["model"], "Model")
if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES:
raise ValueError("Request exceeds 2 MiB; no input is truncated")
state = _content(request["state"], "State")
questions = request["questions"]
if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS:
raise ValueError("Provide 1..128 named questions")
rows = []
for index, (qid, item) in enumerate(questions.items()):
_identifier(qid, "Question ID")
if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"}
or not {"type", "instructions"} <= set(item)):
raise ValueError(qid + ": use type, instructions and optional criteria")
kind = item["type"]
if kind not in ("noul", "choice", "score"):
raise ValueError(qid + ": type must be noul, choice or score")
q = {"id": qid, "type": kind.capitalize(),
"text": _content(item["instructions"], qid + ".instructions")}
criteria = item.get("criteria")
if kind == "choice":
if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255:
raise ValueError(qid + ": Choice requires 2..255 named options")
q["options"] = []
for name, description in criteria.items():
_identifier(name, "Choice option")
text = name if description is None else name + ": " + _content(description, qid + ".criteria")
q["options"].append({"id": name, "text": text})
elif kind == "score":
if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
raise ValueError(qid + ": Score requires 2..10 ordered levels")
q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")}
for i, v in enumerate(criteria)]
elif "criteria" in item:
if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
raise ValueError(qid + ": Noul criteria accept false and true only")
for key in ("false", "true"):
if key in criteria:
q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key)
rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q})
return rows
def _answer(row, prediction):
q = row["question"]
kind = q["type"].lower()
ids = (["no", "yes"] if kind == "noul" else
[v["id"] for v in q["options" if kind == "choice" else "levels"]])
if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"]
or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids
or type(prediction.get("input_tokens")) is not int
or not 1 <= prediction["input_tokens"] <= 1024
or type(prediction.get("state_tokens_original")) is not int
or prediction["state_tokens_original"] < 0
or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")):
raise RuntimeError("Prediction identity or complete-input profile mismatch")
p = prediction.get("probabilities")
if (not isinstance(p, list) or len(p) != len(ids)
or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p)
or abs(sum(p) - 1) > 2e-5):
raise RuntimeError("Invalid prediction probabilities")
answer = {"type": kind}
if kind == "noul":
if prediction.get("probability") != p[1]:
raise RuntimeError("Native Noul probability mismatch")
answer["noul"] = p[1]
return answer
best = ids[max(range(len(p)), key=p.__getitem__)]
if prediction.get("choice_id") != best or prediction.get("confidence") != max(p):
raise RuntimeError("Native Choice/confidence mismatch")
answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"])
if kind == "choice":
answer["choice"] = best
else:
score = prediction.get("score")
if (type(score) not in (int, float) or not math.isfinite(score)
or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5):
raise RuntimeError("Native ordinal Score mismatch")
# Preserve native FP32 arithmetic, not a new CPU reduction.
answer["score"] = score
answer["legend"] = {v["id"]: v["text"] for v in q["levels"]}
return answer
class SystemOne:
"""Local System One API for a loaded Kai, Lex or compatible fine-tune.
evaluate(request) accepts the HTTP body shape; system_one(**request) is its
Python equivalent. batch(requests) flattens independent states into GPU
batches and restores the original request/question order. Default B8 groups rows by decision type;
batching='auto' opts into the published homogeneous padding-aware B32 path.
"""
def __init__(self, native, *, model=None, batching="default"):
if model is None:
model = next((name for name, sha in PUBLIC_MODELS.items()
if sha == native.manifest_sha256), None)
_identifier(model, "Model (required for a custom fine-tune)")
# Do not let a different loaded checkpoint claim a published identity.
if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]:
raise ValueError("Loaded checkpoint does not match the public model name")
if batching not in ("default", "auto"):
raise ValueError("batching must be default or auto")
self.native, self.model, self.batching = native, model, batching
def system_one(self, *, state, questions, model=None):
return self.evaluate({"model": self.model if model is None else model,
"state": state, "questions": questions})
def evaluate(self, request):
return self.batch([request])[0]
def batch(self, requests):
if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS:
raise ValueError("Provide 1..128 request objects")
if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES:
raise ValueError("Combined request exceeds 2 MiB")
# Detach mutable caller inputs before conversion/admission/inference.
requests = copy.deepcopy(requests)
groups = []
for request in requests:
rows = system_one_records(request)
if request["model"] != self.model:
raise ValueError("Request model does not match this loaded model")
groups.append(rows)
count = sum(map(len, groups))
if count > MAX_DECISIONS:
raise ValueError("Provide at most 512 decisions in one batch")
records, slots = [], []
# Question-major order permits the same question across many states to
# share a physical batch; external IDs never decide caching or grouping.
for qi in range(max(map(len, groups))):
for ri, group in enumerate(groups):
if qi < len(group):
row = group[qi]
row["id"] = f"systemone:{ri}:{qi}"
records.append(row)
slots.append((ri, qi))
if self.batching == "auto":
from ._auto import predict_auto_1k
predictions = predict_auto_1k(self.native, records)
else:
from ._grouped import predict_grouped_1k
predictions = predict_grouped_1k(self.native, records, batch_size=8)
if len(predictions) != len(records):
raise RuntimeError("Incomplete model result; no partial answers returned")
values = [[None] * len(group) for group in groups]
tokens = [0] * len(groups)
for row, prediction, (ri, qi) in zip(records, predictions, slots):
values[ri][qi] = _answer(row, prediction)
tokens[ri] += prediction["input_tokens"]
return [{"model": self.model,
"answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])},
"usage": {"input_tokens": tokens[ri], "output_tokens": 0}}
for ri, group in enumerate(groups)]
|