"""decider-0.8b — typed decisions with calibrated probabilities, in one forward pass. The model never generates text. It reads a *state* (a string or any JSON value) plus one or more *typed questions* (Choice / Noul / Score, each with its own option list) and returns a calibrated probability distribution per question, read off the hidden state at one answer slot per question. The scoring path below is the authors' own reference implementation, vendored verbatim in `decider/` (the inference subset shipped inside https://huggingface.co/Mapika/decider-0.8b, Apache-2.0: `prompt.py`, `model.py`, `systemone.py`, `infer.py`). `_system_one` is `decider.infer.Decider.system_one`'s eager branch with two changes only: * the checkpoint is loaded once at module scope and moved to CUDA eagerly (ZeroGPU), and * the temperature is an explicit argument instead of instance state, so concurrent requests on one worker cannot race each other. `decider.engine` (shape-bucketed CUDA graphs) is deliberately not used: graph capture inside a forked ZeroGPU worker buys little for a single interactive request. Everything else — the prompt layout, the per-question row isolation, the isolated Score levels, the restriction of the answer logits to the option-label tokens and the `-inf` masking of unused slots — is unmodified. """ import os os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") import spaces # noqa: E402 — must precede torch / transformers import json # noqa: E402 import re # noqa: E402 import time # noqa: E402 import gradio as gr # noqa: E402 import torch # noqa: E402 from huggingface_hub import hf_hub_download # noqa: E402 from decider.infer import Example, Q, neutralize_options # noqa: E402 from decider.model import DecisionModel, collate # noqa: E402 from decider.prompt import MAX_OPTIONS, build # noqa: E402 from decider.systemone import ( # noqa: E402 MAX_CHOICE, MAX_LEVELS, assemble, plan_rows, render_question, render_state, unique_tokens, ) MODEL_ID = "Mapika/decider-0.8b" CFG = json.load(open(hf_hub_download(MODEL_ID, "decider_config.json"))) VERSION = CFG.get("version", "dev") MODEL_NAME = "decider-" + str(VERSION) DEFAULT_T = float(CFG.get("temperature", 1.0)) # 1.03 for 0.8b-v1 — the fitted, calibrated value NEUTRALIZE_NONE = bool(CFG.get("neutralize_none", True)) # False for 0.8b-v1 DEFAULT_ISOLATED = bool(CFG.get("isolated_levels", False)) MAX_STATE_TOKENS = int(CFG.get("max_state_tokens", 32768)) MAX_FWD_TOKENS = 65536 # -------------------------------------------------------------------------------------------- # Model — loaded once at module scope, eagerly moved to the (ZeroGPU-intercepted) CUDA device. # -------------------------------------------------------------------------------------------- MODEL = DecisionModel(MODEL_ID, dtype=torch.bfloat16, grad_ckpt=False).to("cuda").eval() TOK = MODEL.tok class _Keep: """The reference's rng stand-in: keep the caller's option order, never shuffle.""" def shuffle(self, x): pass def sample(self, xs, k): return xs[:k] def _system_one(state, questions, temperature, independent, isolated): """decider.infer.Decider.system_one, eager branch, with an explicit temperature.""" ctx = render_state(state) rqs = {k: render_question(v) for k, v in questions.items()} opts = (lambda r: neutralize_options(r["options"])[0]) if NEUTRALIZE_NONE else (lambda r: list(r["options"])) flat, index = plan_rows(rqs, isolated and independent) rows = [[r] for r in flat] if independent else [flat] items = [ build( Example(ctx, [Q(r["question"], opts(r), 0) for r in row]), TOK, _Keep(), max_options=MAX_OPTIONS, max_ctx_tokens=MAX_STATE_TOKENS, layout="state_first", ) for row in rows ] probs = [] with torch.no_grad(): per = max(1, MAX_FWD_TOKENS // max(len(it["ids"]) for it in items)) for i in range(0, len(items), per): chunk = items[i : i + per] bt = collate(chunk, TOK.pad_token_id) lg = MODEL.slot_logits( *[bt[k].to("cuda") for k in ("input_ids", "attention_mask", "slot_idx", "slot_batch", "nopts")] ) pr = torch.softmax(lg / float(temperature), -1).float().cpu() c = 0 for it in chunk: probs.append(pr[c : c + len(it["slots"])]) c += len(it["slots"]) flatp = [p.tolist() for ps in probs for p in ps] return ( { "model": MODEL_NAME, "answers": assemble(rqs, index, flatp), "usage": {"input_tokens": unique_tokens(items), "output_tokens": 0}, }, items, ) # -------------------------------------------------------------------------------------------- # The question mini-format -> the Jev-shaped question dict the reference expects # -------------------------------------------------------------------------------------------- TYPES = ("choice", "noul", "bool", "score") _ID_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_.-]*$") class ParseError(ValueError): pass def _maybe_json(text): """Option descriptions may be a string or any JSON value (the reference renders both).""" t = text.strip() if t[:1] in "{[" or t in ("null", "true", "false"): try: return json.loads(t) except Exception: return text return text def parse_questions(block: str) -> dict: """Parse the textarea into `{id: {"type", "instructions", "criteria"}}`. One question per blank-line-separated paragraph: : -