"""The plain lettered prompt an untuned base is read through as a direct-logit classifier (JevBench's raw-base rows).
One module for both readers, so the text is the same byte for byte:
eval.predictors.RawLogitPredictor scores Kev-format records (it renders them with kev.api.to_record, then uses
raw_options / raw_prompt / prompt_ids / letter_token_ids from here)
midtrain decision anchor (train.decision_anchor_weight) renders traj2model training records with
record_prompts, which gives the same text as RawLogitPredictor on
t2m_kev.convert.to_kev_request(record) (eval/tests/test_eval_rawlogit.py checks it)
Template (RAW_PROMPT_VERSION):
State:
{state} (the rendered state; the block is left out when it is blank)
Question: {instructions}
A. {option 1}
B. {option 2}
...
Answer:
`<|name|>` in caller text becomes `<¦name¦>` (Kev's rule). No chat template, no trailing space after "Answer:"; the ids
are tokenizer(prompt, add_special_tokens=True). Readout: z_L = log(P("L") + P(" L")) over the full next-token
distribution, p = softmax over the present options' z. No torch or kev import here: eval imports it at module load.
A second layout, SEMIF_CHAT (semif_prompt, "semif_chat/1"), is JevK5's / Jobe's prompt byte for byte (SemIf's protocol,
TheoLeeCJ/SemIf, MIT): a system instruction, the question as one JSON object {evidence, criterion, options: [{letter,
description}]} in the user turn, the Qwen chat template with thinking off, answer letters A-P. See the section below.
"""
import json
import re
LETTERS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
RAW_PROMPT_VERSION = "rawlogit/1"
_SPECIAL = re.compile(r"<\|([A-Za-z0-9_]+)\|>") # Kev's rule (midtrain.encode.user_tokens): caller text never forms <|special|>
POSITIONAL = re.compile(r"[a-z]|o[0-9]+") # choice keys that only number the options (kev.api / encode.option_texts)
_SLUG = re.compile(r"[a-z0-9][a-z0-9_.-]{0,39}") # t2m_kev.convert.SLUG / encode._SLUG
class TooManyOptions(ValueError):
"""More options than letters: the letter readout cannot score the question."""
def plain(text):
return _SPECIAL.sub(r"<¦\1¦>", str(text))
def raw_options(qtype, keys, texts):
"""The option strings the base sees after its letter: the rendered option texts (kev.api.to_record's, the strings
midtrain.encode.option_texts gives for a traj2model question), except that a choice question whose keys are all
positional (a, b, c, ... / o27) drops the `a: ` prefix, since the letter replaces it."""
if qtype == "choice":
keys = list(keys)
if all(POSITIONAL.fullmatch(k) for k in keys):
return [t[len(k) + 2:] if t.startswith(f"{k}: ") else t for k, t in zip(keys, texts)]
return list(texts)
def raw_prompt(state, instructions, options):
"""The plain prompt (module docstring). `state` and `instructions` are already rendered text."""
if len(options) > len(LETTERS):
raise TooManyOptions(f"{len(options)} options: the letter readout has {len(LETTERS)} letters")
head = f"State:\n{plain(state)}\n\n" if str(state).strip() else ""
lines = [f"{LETTERS[j]}. {plain(o)}" for j, o in enumerate(options)]
return head + f"Question: {plain(instructions)}\n" + "\n".join(lines) + "\nAnswer:"
def prompt_ids(tok, prompt):
"""The token ids RawLogitPredictor feeds the base for a prompt."""
return tok(prompt, add_special_tokens=True).input_ids
def letter_token_ids(tok, letters=LETTERS):
"""{letter: [token ids]}: the single-token encodings of "A" and " A" (both, deduplicated; a variant that is not one
token is left out). A letter with neither is an error: the readout would have nothing to read."""
out = {}
for L in letters:
ids = []
for v in (L, " " + L):
enc = tok(v, add_special_tokens=False).input_ids
if len(enc) == 1 and enc[0] not in ids:
ids.append(enc[0])
if not ids:
raise ValueError(f"tokenizer has no single-token encoding of {L!r} or {' ' + L!r}")
out[L] = ids
return out
def choice_keys(options):
"""A traj2model choice question's Kev keys (t2m_kev.convert.option_keys): the options themselves when they are all
distinct short slugs, else a, b, c, ... (o1, o2, ... past 26)."""
opts = [str(o) for o in options]
if opts and len(set(opts)) == len(opts) and all(isinstance(o, str) and _SLUG.fullmatch(o) for o in options):
return opts
return [chr(ord("a") + i) if len(opts) <= 26 else f"o{i + 1}" for i in range(len(opts))]
def question_prompt(state_text, q):
"""The raw prompt of one traj2model question under an already rendered state; TooManyOptions past 26 options."""
from .encode import option_texts # torch-importing module; only the training side calls this
texts = option_texts(q)
keys = choice_keys(q["options"]) if q["type"] == "choice" else None
return raw_prompt(state_text, str(q["instructions"]), raw_options(q["type"], keys, texts))
def record_prompts(rec):
"""traj2model decision record -> [prompt or None] per question, in question order: the text RawLogitPredictor scores
for t2m_kev.convert.to_kev_request(rec). None where the question has more options than letters."""
from .encode import render
state = render(rec["state"])
out = []
for q in rec["questions"]:
try:
out.append(question_prompt(state, q))
except TooManyOptions:
out.append(None)
return out
def attach_raw_ids(encs, rec, tok, max_tokens):
"""Decision anchor, data side (the prefetch thread): give each decision encoding of `rec` two per-branch lists,
`raw_ids` (the prompt ids under `tok`, the reference model's tokenizer, or None) and `raw_skip` (None when the
branch is anchored; "tier2" for a rationale branch, which repeats its question and is not a candidate; "length" when
the prompt exceeds max_tokens; "options" past 26 options). Each question is rendered and tokenized once."""
prompts, cache = None, {}
for e in encs:
ids, skip = [], []
for qi, tier in zip(e["qidx"], e["tiers"]):
if tier == 2:
ids.append(None); skip.append("tier2")
continue
if qi not in cache:
if prompts is None:
prompts = record_prompts(rec)
p = prompts[qi]
if p is None:
cache[qi] = (None, "options")
else:
x = prompt_ids(tok, p)
cache[qi] = (x, None) if len(x) <= int(max_tokens) else (None, "length")
ids.append(cache[qi][0]); skip.append(cache[qi][1])
e["raw_ids"], e["raw_skip"] = ids, skip
return encs
# --- semif_chat: JevK5 / Jobe's prompt (docs/hard-tier-leaders.md technique 1) ----------------------------------------
# Reproduced from, byte for byte (trainer/eval/tests/test_eval_semif_chat.py checks it against prompts rendered by their
# code at these commits, trainer/eval/tests/fixtures/semif_chat_prompts.json):
# allebee/jevk5 @7b97499 jevk5/prompt.py:13 (LETTERS), 21-24 (SYSTEM), 28-32 (CHAT_TEMPLATE, "verified against
# apply_chat_template(..., add_generation_prompt=True, enable_thinking=False)"), 35-50 (messages / prompt_text),
# 53-65 (decision_options)
# MantisShrimpdev/jobe @0382f20 src/jobe/prompt.py:20-23, 95-109, 142-166; src/jobe/server.py:99-117 (options_for),
# 199-201 (Decision(evidence=state, criterion=instructions)); src/jobe/slots.py:83-128 (letter ids resolved in context)
# Jobe's server additionally describes index-only choice options from a jev-browser page state (server.py:159-175,
# describe_indices); it only fires on choice criteria with EMPTY descriptions and a state carrying page.elements, which
# no JevBench / transfer / sealed-like item has, so it is left out (JevK5 does not have it either).
# Unlike the plain layout: no `<|name|>` rewrite (their JSON carries caller text as is), the state is the request's JSON
# value (not kev.api.render's text), noul reads A = true, B = false, and at most 16 options (letters A-P).
SEMIF_PROMPT_VERSION = "semif_chat/1"
SEMIF_LETTERS = "ABCDEFGHIJKLMNOP"
SEMIF_SYSTEM = ("Apply the supplied criterion to the supplied evidence. Choose exactly one listed option. "
"Respond with only its uppercase letter, with no explanation or reasoning.")
SEMIF_CHAT_TEMPLATE = ("<|im_start|>system\n{system}<|im_end|>\n"
"<|im_start|>user\n{user}<|im_end|>\n"
"<|im_start|>assistant\n\n\n\n\n")
def semif_options(qtype, criteria):
"""[(option id, description)] of a Kev/TypeSafe question (type, criteria as sent on the wire), in the order the letters
are given: noul true, false (the criteria text, else "The proposition is ."); choice the criteria keys (the value,
else the key); score the level indices. Descriptions are ": " (JevK5 decision_options, Jobe options_for)."""
if qtype == "noul":
c = criteria or {}
pairs = [(k, c.get(k) or f"The proposition is {k}.") for k in ("true", "false")]
elif qtype == "choice":
c = dict.fromkeys(criteria) if isinstance(criteria, list) else (criteria or {})
pairs = [(str(k), v or str(k)) for k, v in c.items()]
elif qtype == "score":
pairs = [(str(i), level) for i, level in enumerate(criteria or [])]
else:
raise ValueError(f"unknown question type {qtype!r}")
return [(k, f"{k}: {d}") for k, d in pairs]
def semif_messages(state, criterion, descriptions):
"""The two chat messages (system, user JSON) for one question; `state` is the request's state value (any JSON)."""
if len(descriptions) > len(SEMIF_LETTERS):
raise TooManyOptions(f"{len(descriptions)} options: the semif_chat readout has {len(SEMIF_LETTERS)} letters")
payload = {"evidence": state, "criterion": criterion,
"options": [{"letter": SEMIF_LETTERS[i], "description": d} for i, d in enumerate(descriptions)]}
return [{"role": "system", "content": SEMIF_SYSTEM}, {"role": "user", "content": json.dumps(payload, ensure_ascii=False)}]
def semif_prompt(state, criterion, descriptions):
"""The full semif_chat prompt text, chat template included and ending after `\n\n\n\n`, where the answer
letter goes. Tokenized with add_special_tokens=False (the template carries its own special tokens)."""
system, user = semif_messages(state, criterion, descriptions)
return SEMIF_CHAT_TEMPLATE.format(system=system["content"], user=user["content"])
def semif_question_prompt(state, question):
"""(prompt, [option id in letter order]) for one Kev/TypeSafe question dict (type, instructions, criteria)."""
opts = semif_options(question["type"], question.get("criteria"))
return semif_prompt(state, question.get("instructions", ""), [d for _, d in opts]), [k for k, _ in opts]
def semif_letter_ids(tok, prompt, ids, count):
"""The token id of each of the first `count` answer letters, resolved in context as Jobe does (slots.py:83-128):
encode(prompt + L)[-1], checking that the letter is one token and does not re-tokenize the prompt's tail."""
out = []
for L in SEMIF_LETTERS[:count]:
merged = tok(prompt + L, add_special_tokens=False).input_ids
if len(merged) != len(ids) + 1 or merged[:-1] != ids:
raise ValueError(f"answer letter {L!r} is not one clean token after the semif_chat prompt")
out.append(merged[-1])
if len(set(out)) != len(out):
raise ValueError("answer letters collide in context")
return out