Download evaluation/code/swift15/ifbench_score.py from daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen: direct link, hf CLI and curl.
- Browser
- Download file 876 Bytes
-
https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/ifbench_score.py
- Command line
-
hf download hf://daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/evaluation/code/swift15/ifbench_score.py
-
curl -L -o ifbench_score.py https://huggingface.co/daavidhauser/Swift-1.5-Qwen3.8-27B-W4A16-HyperQwen/resolve/main/evaluation/code/swift15/ifbench_score.py
876 Bytes
| """Official IFBench strict prompt-level scoring in the isolated eval environment.""" | |
| import json | |
| import sys | |
| from ifbench import instructions_registry | |
| def score(task, response): | |
| passed = [] | |
| for name, kwargs in zip(task["instruction_id_list"], task["kwargs"]): | |
| checker = instructions_registry.INSTRUCTION_DICT[name](name) | |
| checker.build_description(**{k: v for k, v in kwargs.items() if v is not None}) | |
| needed = checker.get_instruction_args() | |
| if needed and "prompt" in needed: | |
| checker.build_description(prompt=task["messages"][-1]["content"]) | |
| passed.append(bool(response.strip()) and bool(checker.check_following(response))) | |
| return {"correct": all(passed), "instructions": passed} | |
| if __name__ == "__main__": | |
| rows = json.load(sys.stdin) | |
| print(json.dumps([score(r["task"], r["response"]) for r in rows])) | |