Download decision_inference/_system_one.py from vllm-sr/Decision-1.0-Lex-0.6B: direct link, hf CLI and curl.
- Browser
- Download file 10.7 kB
-
https://huggingface.co/vllm-sr/Decision-1.0-Lex-0.6B/resolve/e4a96e95ad08867f7114e1a3ed4d42dca4e2723e/decision_inference/_system_one.py
- Command line
-
hf download hf://vllm-sr/Decision-1.0-Lex-0.6B@e4a96e95ad08867f7114e1a3ed4d42dca4e2723e/decision_inference/_system_one.py
-
curl -L -o _system_one.py https://huggingface.co/vllm-sr/Decision-1.0-Lex-0.6B/resolve/e4a96e95ad08867f7114e1a3ed4d42dca4e2723e/decision_inference/_system_one.py
10.7 kB
| """System One request/answer schema over complete-input Decision inference. | |
| Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping; | |
| Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache. | |
| """ | |
| import copy | |
| import json | |
| import math | |
| MAX_QUESTIONS = 128 | |
| MAX_REQUESTS = 128 | |
| MAX_DECISIONS = 512 | |
| MAX_REQUEST_BYTES = 2 * 1024 * 1024 | |
| PUBLIC_MODELS = { | |
| "Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe", | |
| "Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6", | |
| } | |
| def _identifier(value, label): | |
| if not isinstance(value, str) or not value.strip() or len(value) > 128: | |
| raise ValueError(label + " must be a nonempty string of at most 128 characters") | |
| return value | |
| def _json(value): | |
| # JSON objects must have string keys: never silently coerce Python keys. | |
| def check(item): | |
| if isinstance(item, dict): | |
| if not all(isinstance(k, str) for k in item): | |
| raise ValueError("JSON object keys must be strings") | |
| for v in item.values(): | |
| check(v) | |
| elif isinstance(item, list): | |
| for v in item: | |
| check(v) | |
| elif item is not None and not isinstance(item, (str, bool, int, float)): | |
| raise ValueError("Only JSON values are supported") | |
| try: | |
| check(value) | |
| return json.dumps(value, ensure_ascii=False, sort_keys=True, | |
| separators=(",", ":"), allow_nan=False) | |
| except (TypeError, RecursionError, UnicodeError) as exc: | |
| raise ValueError("Invalid JSON content") from exc | |
| def _content(value, label): | |
| if isinstance(value, str): | |
| if not value.strip(): | |
| raise ValueError(label + " must not be empty") | |
| return value | |
| if isinstance(value, (dict, list)): | |
| return _json(value) | |
| raise ValueError(label + " must be text, an object, or an array") | |
| def system_one_records(request): | |
| """Validate one wire request and return native rows, without loading a model. | |
| Full token admission occurs in predict_1k before the first model forward. | |
| External record IDs should be made unique when combining training examples. | |
| """ | |
| if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}: | |
| raise ValueError("A request contains exactly model, state and questions") | |
| _identifier(request["model"], "Model") | |
| if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES: | |
| raise ValueError("Request exceeds 2 MiB; no input is truncated") | |
| state = _content(request["state"], "State") | |
| questions = request["questions"] | |
| if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS: | |
| raise ValueError("Provide 1..128 named questions") | |
| rows = [] | |
| for index, (qid, item) in enumerate(questions.items()): | |
| _identifier(qid, "Question ID") | |
| if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"} | |
| or not {"type", "instructions"} <= set(item)): | |
| raise ValueError(qid + ": use type, instructions and optional criteria") | |
| kind = item["type"] | |
| if kind not in ("noul", "choice", "score"): | |
| raise ValueError(qid + ": type must be noul, choice or score") | |
| q = {"id": qid, "type": kind.capitalize(), | |
| "text": _content(item["instructions"], qid + ".instructions")} | |
| criteria = item.get("criteria") | |
| if kind == "choice": | |
| if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255: | |
| raise ValueError(qid + ": Choice requires 2..255 named options") | |
| q["options"] = [] | |
| for name, description in criteria.items(): | |
| _identifier(name, "Choice option") | |
| text = name if description is None else name + ": " + _content(description, qid + ".criteria") | |
| q["options"].append({"id": name, "text": text}) | |
| elif kind == "score": | |
| if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10: | |
| raise ValueError(qid + ": Score requires 2..10 ordered levels") | |
| q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")} | |
| for i, v in enumerate(criteria)] | |
| elif "criteria" in item: | |
| if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}: | |
| raise ValueError(qid + ": Noul criteria accept false and true only") | |
| for key in ("false", "true"): | |
| if key in criteria: | |
| q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key) | |
| rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q}) | |
| return rows | |
| def _answer(row, prediction): | |
| q = row["question"] | |
| kind = q["type"].lower() | |
| ids = (["no", "yes"] if kind == "noul" else | |
| [v["id"] for v in q["options" if kind == "choice" else "levels"]]) | |
| if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"] | |
| or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids | |
| or type(prediction.get("input_tokens")) is not int | |
| or not 1 <= prediction["input_tokens"] <= 1024 | |
| or type(prediction.get("state_tokens_original")) is not int | |
| or prediction["state_tokens_original"] < 0 | |
| or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")): | |
| raise RuntimeError("Prediction identity or complete-input profile mismatch") | |
| p = prediction.get("probabilities") | |
| if (not isinstance(p, list) or len(p) != len(ids) | |
| or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p) | |
| or abs(sum(p) - 1) > 2e-5): | |
| raise RuntimeError("Invalid prediction probabilities") | |
| answer = {"type": kind} | |
| if kind == "noul": | |
| if prediction.get("probability") != p[1]: | |
| raise RuntimeError("Native Noul probability mismatch") | |
| answer["noul"] = p[1] | |
| return answer | |
| best = ids[max(range(len(p)), key=p.__getitem__)] | |
| if prediction.get("choice_id") != best or prediction.get("confidence") != max(p): | |
| raise RuntimeError("Native Choice/confidence mismatch") | |
| answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"]) | |
| if kind == "choice": | |
| answer["choice"] = best | |
| else: | |
| score = prediction.get("score") | |
| if (type(score) not in (int, float) or not math.isfinite(score) | |
| or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5): | |
| raise RuntimeError("Native ordinal Score mismatch") | |
| # Preserve native FP32 arithmetic, not a new CPU reduction. | |
| answer["score"] = score | |
| answer["legend"] = {v["id"]: v["text"] for v in q["levels"]} | |
| return answer | |
| class SystemOne: | |
| """Local System One API for a loaded Kai, Lex or compatible fine-tune. | |
| evaluate(request) accepts the HTTP body shape; system_one(**request) is its | |
| Python equivalent. batch(requests) flattens independent states into GPU | |
| batches and restores the original request/question order. Default B8 groups rows by decision type; | |
| batching='auto' opts into the published homogeneous padding-aware B32 path. | |
| """ | |
| def __init__(self, native, *, model=None, batching="default"): | |
| if model is None: | |
| model = next((name for name, sha in PUBLIC_MODELS.items() | |
| if sha == native.manifest_sha256), None) | |
| _identifier(model, "Model (required for a custom fine-tune)") | |
| # Do not let a different loaded checkpoint claim a published identity. | |
| if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]: | |
| raise ValueError("Loaded checkpoint does not match the public model name") | |
| if batching not in ("default", "auto"): | |
| raise ValueError("batching must be default or auto") | |
| self.native, self.model, self.batching = native, model, batching | |
| def system_one(self, *, state, questions, model=None): | |
| return self.evaluate({"model": self.model if model is None else model, | |
| "state": state, "questions": questions}) | |
| def evaluate(self, request): | |
| return self.batch([request])[0] | |
| def batch(self, requests): | |
| if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS: | |
| raise ValueError("Provide 1..128 request objects") | |
| if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES: | |
| raise ValueError("Combined request exceeds 2 MiB") | |
| # Detach mutable caller inputs before conversion/admission/inference. | |
| requests = copy.deepcopy(requests) | |
| groups = [] | |
| for request in requests: | |
| rows = system_one_records(request) | |
| if request["model"] != self.model: | |
| raise ValueError("Request model does not match this loaded model") | |
| groups.append(rows) | |
| count = sum(map(len, groups)) | |
| if count > MAX_DECISIONS: | |
| raise ValueError("Provide at most 512 decisions in one batch") | |
| records, slots = [], [] | |
| # Question-major order permits the same question across many states to | |
| # share a physical batch; external IDs never decide caching or grouping. | |
| for qi in range(max(map(len, groups))): | |
| for ri, group in enumerate(groups): | |
| if qi < len(group): | |
| row = group[qi] | |
| row["id"] = f"systemone:{ri}:{qi}" | |
| records.append(row) | |
| slots.append((ri, qi)) | |
| if self.batching == "auto": | |
| from ._auto import predict_auto_1k | |
| predictions = predict_auto_1k(self.native, records) | |
| else: | |
| from ._grouped import predict_grouped_1k | |
| predictions = predict_grouped_1k(self.native, records, batch_size=8) | |
| if len(predictions) != len(records): | |
| raise RuntimeError("Incomplete model result; no partial answers returned") | |
| values = [[None] * len(group) for group in groups] | |
| tokens = [0] * len(groups) | |
| for row, prediction, (ri, qi) in zip(records, predictions, slots): | |
| values[ri][qi] = _answer(row, prediction) | |
| tokens[ri] += prediction["input_tokens"] | |
| return [{"model": self.model, | |
| "answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])}, | |
| "usage": {"input_tokens": tokens[ri], "output_tokens": 0}} | |
| for ri, group in enumerate(groups)] | |