"""Official IFBench strict prompt-level scoring in the isolated eval environment.""" import json import sys from ifbench import instructions_registry def score(task, response): passed = [] for name, kwargs in zip(task["instruction_id_list"], task["kwargs"]): checker = instructions_registry.INSTRUCTION_DICT[name](name) checker.build_description(**{k: v for k, v in kwargs.items() if v is not None}) needed = checker.get_instruction_args() if needed and "prompt" in needed: checker.build_description(prompt=task["messages"][-1]["content"]) passed.append(bool(response.strip()) and bool(checker.check_following(response))) return {"correct": all(passed), "instructions": passed} if __name__ == "__main__": rows = json.load(sys.stdin) print(json.dumps([score(r["task"], r["response"]) for r in rows]))