daavidhauser's picture
Publish Swift HyperQwen collection with performance and quality comparisons
2bc6021 verified
Raw History Blame Contribute Delete
876 Bytes
"""Official IFBench strict prompt-level scoring in the isolated eval environment."""
import json
import sys
from ifbench import instructions_registry
def score(task, response):
passed = []
for name, kwargs in zip(task["instruction_id_list"], task["kwargs"]):
checker = instructions_registry.INSTRUCTION_DICT[name](name)
checker.build_description(**{k: v for k, v in kwargs.items() if v is not None})
needed = checker.get_instruction_args()
if needed and "prompt" in needed:
checker.build_description(prompt=task["messages"][-1]["content"])
passed.append(bool(response.strip()) and bool(checker.check_following(response)))
return {"correct": all(passed), "instructions": passed}
if __name__ == "__main__":
rows = json.load(sys.stdin)
print(json.dumps([score(r["task"], r["response"]) for r in rows]))