# Zero-Token Confidence (ZTC) — usage for Darwin-180B-RSI # # ZTC reads the model's own internal state ONCE, before generation, and returns # the probability that the answer the model is about to produce will be correct. # No extra tokens are generated. No second model is required. # # probe file : ztc/ztc_probe_darwin180rsi.npz # input : final-layer hidden state of the last prompt token (2560-dim) # output : score, and a calibrated probability in [0, 1] import json import numpy as np import torch from transformers import AutoModelForImageTextToText, AutoProcessor MODEL = "FINAL-Bench/Darwin-180B-RSI" PROBE = "ztc/ztc_probe_darwin180rsi.npz" class ZTC: def __init__(self, path=PROBE): z = np.load(path) self.w, self.mu, self.sd = z["w"].astype(np.float32), z["mu"].astype(np.float32), z["sd"].astype(np.float32) self.s_mean, self.s_std = float(z["s_mean"]), float(z["s_std"]) self.A, self.B = float(z["cal_A"]), float(z["cal_B"]) def score(self, hidden): s = ((np.asarray(hidden, np.float32) - self.mu) / self.sd) @ self.w p = 1.0 / (1.0 + np.exp(-(self.A * (s - self.s_mean) / self.s_std + self.B))) return float(s), float(p) proc = AutoProcessor.from_pretrained(MODEL) model = AutoModelForImageTextToText.from_pretrained(MODEL, torch_dtype=torch.bfloat16, device_map="auto").eval() ztc = ZTC() question = "What is 17 * 23?" text = proc.apply_chat_template([{"role": "user", "content": [{"type": "text", "text": question}]}], add_generation_prompt=True, tokenize=False) enc = proc(text=[text], return_tensors="pt").to(model.device) with torch.no_grad(): h = model(**enc, output_hidden_states=True, use_cache=False).hidden_states[-1][0, -1].float().cpu().numpy() s, p = ztc.score(h) print(json.dumps({"ztc_score": round(s, 4), "confidence": round(p, 4)})) # Gate the action, not the answer: # if p < THRESHOLD: do not call the tool / escalate / answer "I don't know" # else: generate as usual