alpha-sys-1-3B / alpha_sys_1.py
marcodsn's picture
Upload folder using huggingface_hub
a790235 verified
Raw History Blame
8.7 kB
"""alpha-sys-1 inference client: one file, no dependency on this repository, shipped in the
Hugging Face repos as `alpha_sys_1.py`. It renders questions exactly as the model was trained
on them and reads the answer distribution from one forward pass.
from alpha_sys_1 import SystemOne
m = SystemOne("nullsilver/alpha-sys-1-1.6B")
m.ask({"type": "choice", "instructions": "Which team should handle this?",
"criteria": {"billing": "payments, refunds", "technical": "bugs, outages", "sales": "pricing"}},
state="Our API started returning 500 errors this morning.")
# -> {"choice": "technical", "probabilities": {...}, "confidence": 0.71}
m.system_one({"state": ..., "images": [...], "questions": {"q1": {...}, "q2": {...}}})
# -> {"model": ..., "answers": {"q1": {...}, "q2": {...}}} (TypeSafe's System One shape)
Question types: choice (criteria = {option: description or None} or a list of options, up to
26), noul (a statement; criteria = {"true": ..., "false": ...} optional), score (criteria = the
levels, lowest first; the score is the expected level index). Images: a PIL image, a path, or
a data URL; small images are upscaled to 256 px as in training.
"""
from __future__ import annotations
import base64
import io
import math
import string
from typing import Any
import torch
from PIL import Image
from transformers import AutoModelForImageTextToText, AutoProcessor
IMAGE_SIDE = 256
def render_state(state: Any) -> str:
if state is None:
return ""
if isinstance(state, str):
return state
if isinstance(state, dict):
return "\n".join(f"{k}: {v}" for k, v in state.items())
return str(state)
def render(state: Any, q: dict) -> tuple[str, list[str], list[str]]:
"""-> (user text, label tokens in listed order, answer-space keys in the same order)."""
parts = [s for s in [render_state(state)] if s]
t = q["type"]
if t == "noul":
c = q.get("criteria") or {}
clar = "".join(f"\n{lab} means: {c[k]}" for lab, k in (("Yes", "true"), ("No", "false")) if c.get(k))
parts.append(f"Statement: {q['instructions']}{clar}\nIs the statement true? Answer with Yes or No only.")
return "\n\n".join(parts), ["No", "Yes"], ["no", "yes"]
crit = q["criteria"]
if t == "choice":
items = list(crit.items()) if isinstance(crit, dict) else [(o, None) for o in crit]
keys = [k for k, _ in items]
else:
items, keys = [(lvl, None) for lvl in crit], [str(i) for i in range(len(crit))]
if len(items) > 26:
raise ValueError("at most 26 options or levels per question")
labels = list(string.ascii_uppercase[: len(items)])
lines = [f"{lab}. {o}" + (f": {d}" if d else "") for lab, (o, d) in zip(labels, items)]
parts.append(f"{q['instructions']}\n" + "\n".join(lines) + "\nAnswer with the letter only.")
return "\n\n".join(parts), labels, keys
def load_image(im: Any) -> Image.Image:
if isinstance(im, Image.Image):
img = im
elif isinstance(im, str) and im.startswith("data:"):
img = Image.open(io.BytesIO(base64.b64decode(im.split(",", 1)[1])))
else:
img = Image.open(im)
img = img.convert("RGB")
if max(img.size) < IMAGE_SIDE:
img = img.resize((IMAGE_SIDE, IMAGE_SIDE), Image.BICUBIC)
return img
def confidence(p: list[float]) -> float:
n = len(p)
if n < 2:
return 1.0
h = -sum(x * math.log(x) for x in p if x > 0)
return round(max(0.0, 1 - h / math.log(n)), 4)
class SystemOne:
def __init__(self, repo: str, revision: str | None = None, device: str | None = None, dtype=torch.bfloat16, temperature: float = 1.0):
"""temperature: the label logits are divided by it (1.0 = the model as released; see fit_temperature)."""
self.repo, self.revision, self.temperature = repo, revision, temperature
self.processor = AutoProcessor.from_pretrained(repo, revision=revision)
self.processor.tokenizer.padding_side = "left"
self.model = AutoModelForImageTextToText.from_pretrained(repo, revision=revision, dtype=dtype)
self.device = device or ("cuda" if torch.cuda.is_available() else "cpu")
self.model.to(self.device).eval()
self._ids: dict[str, int] = {}
def _label_id(self, label: str) -> int:
if label not in self._ids:
ids = self.processor.tokenizer.encode(label, add_special_tokens=False)
assert len(ids) == 1, label
self._ids[label] = ids[0]
return self._ids[label]
@torch.inference_mode()
def distributions(self, items: list[tuple[Any, dict, list | None]]) -> list[list[float]]:
"""items: (state, question, images or None) -> probabilities in the answer-space order."""
msgs, labels_per = [], []
for state, q, images in items:
text, labels, _ = render(state, q)
content = [{"type": "image", "image": load_image(im)} for im in (images or [])] + [{"type": "text", "text": text}]
msgs.append([{"role": "user", "content": content}])
labels_per.append(labels)
inputs = self.processor.apply_chat_template(
msgs, add_generation_prompt=True, tokenize=True, return_dict=True,
processor_kwargs={"return_tensors": "pt", "padding": True}).to(self.device)
logits = self.model(**inputs, logits_to_keep=1).logits[:, -1].float()
out = []
for i, labels in enumerate(labels_per):
ids = torch.tensor([self._label_id(lab) for lab in labels], device=logits.device)
out.append(torch.softmax(logits[i, ids] / self.temperature, -1).tolist())
return out
def answer(self, q: dict, p: list[float]) -> dict:
_, _, keys = render(None, q)
if q["type"] == "choice":
return {"type": "choice", "choice": keys[max(range(len(p)), key=p.__getitem__)],
"probabilities": dict(zip(keys, p)), "confidence": confidence(p)}
if q["type"] == "noul":
return {"type": "noul", "noul": p[1]}
return {"type": "score", "score": sum(i * x for i, x in enumerate(p)),
"legend": dict(zip(keys, q["criteria"])), "probabilities": p, "confidence": confidence(p)}
def ask(self, q: dict, state: Any = None, images: list | None = None) -> dict:
return self.answer(q, self.distributions([(state, q, images)])[0])
def system_one(self, request: dict, batch: int = 16) -> dict:
"""A request in TypeSafe's System One shape: {state, images?, questions: {id: q}}."""
state, images = request.get("state"), request.get("images")
ids = list(request["questions"])
answers = {}
for s in range(0, len(ids), batch):
chunk = ids[s : s + batch]
ps = self.distributions([(state, request["questions"][i], images) for i in chunk])
for i, p in zip(chunk, ps):
answers[i] = self.answer(request["questions"][i], p)
return {"model": self.repo + (f"@{self.revision}" if self.revision else ""), "answers": answers}
def fit_temperature(model: SystemOne, examples: list[tuple[Any, dict, list | None, int]], batch: int = 16) -> float:
"""One scalar that minimises NLL on labelled examples (state, question, images, index of the
true answer in the answer space: option position, 0/1 for noul, level index for score).
A few hundred examples are enough. Use it as SystemOne(..., temperature=T)."""
old, model.temperature = model.temperature, 1.0
try:
probs, truth = [], []
for s in range(0, len(examples), batch):
chunk = examples[s : s + batch]
probs += model.distributions([(st, q, im) for st, q, im, _ in chunk])
truth += [t for _, _, _, t in chunk]
finally:
model.temperature = old
logs = [[math.log(max(x, 1e-12)) for x in p] for p in probs]
def nll(t: float) -> float:
total = 0.0
for lp, y in zip(logs, truth):
z = [v / t for v in lp]
m = max(z)
total -= z[y] - (m + math.log(sum(math.exp(v - m) for v in z)))
return total / len(logs)
lo, hi = math.log(0.05), math.log(20.0) # golden-section search on log T
g = (math.sqrt(5) - 1) / 2
a, b = hi - g * (hi - lo), lo + g * (hi - lo)
fa, fb = nll(math.exp(a)), nll(math.exp(b))
for _ in range(60):
if fa < fb:
hi, b, fb = b, a, fa
a = hi - g * (hi - lo)
fa = nll(math.exp(a))
else:
lo, a, fa = a, b, fb
b = lo + g * (hi - lo)
fb = nll(math.exp(b))
return round(math.exp((lo + hi) / 2), 3)