"""Version 1: invented inventory facts; disjoint entities, same task templates. This corpus checks wiring and learning only. It is not a generalization benchmark. Each entity yields three questions sharing its state. Candidate order is seeded. """ import random VERSION = "inventory-v1" COLORS = ["red", "blue", "green", "yellow"] def make_split(split, groups, seed): rng = random.Random(seed) examples = [] for i in range(groups): color = COLORS[i % 4] count = (i * 3 + i // 4) % 9 sealed = (i // 2) % 2 == 0 entity = f"{split.upper()}-{i:03d}" state = (f"Inventory record for parcel {entity}. The label is {color}. " f"It contains {count} metal washers. The seal is " f"{'intact' if sealed else 'broken'}. The destination is shelf {(i * 7) % 13}.") tasks = [ ("color", "What color is the parcel label?", COLORS[:], color), ("seal", "Is the parcel seal intact?", ["yes", "no"], "yes" if sealed else "no"), ("quantity", "How many washers does the parcel contain?", ["none", "one to four", "five or more"], "none" if count == 0 else "one to four" if count < 5 else "five or more"), ] for task, question, choices, correct in tasks: rng.shuffle(choices) examples.append({"id": f"{entity}-{task}", "group": entity, "family": task, "state": state, "question": question, "choices": choices, "target": choices.index(correct)}) rng.shuffle(examples) return examples def make_data(): return {"train": make_split("train", 64, 101), "calibration": make_split("calibration", 16, 202), "test": make_split("test", 24, 303)}