Agnes-3.0-Flash-NVFP4 / reproducibility /select_longbench.py
ProCreations's picture
Pin matched Agnes native validation suite
63f6cfd verified
Raw History Blame
2.49 kB
"""Select a reproducible, domain-balanced LongBench v2 diagnostic subset."""
import hashlib
import json
import random
from pathlib import Path
from huggingface_hub import hf_hub_download
from transformers import AutoTokenizer
ROOT=Path(__file__).resolve().parent
REV='2b48e494f2c7a2f0af81aae178e05c7e1dde0fe9'
SOURCE_REV='87420286149d9cce9bd46cd335ef9bda33c37c1b'
def prompt_for(row):
return ('Read the context and answer the question.\n\nContext:\n'+row['context']+
'\n\nQuestion: '+row['question']+'\n'+
'\n'.join(letter+'. '+row['choice_'+letter] for letter in 'ABCD')+
'\nPut your final answer letter inside \\boxed{}.')
def main():
raw=json.loads(Path(hf_hub_download('zai-org/LongBench-v2','data.json',repo_type='dataset',revision=REV)).read_text())
tokenizer=AutoTokenizer.from_pretrained('nex-agi/Nex-N2.5-mini',revision=SOURCE_REV)
rng=random.Random(20260909)
chosen=[]
for domain in sorted({r['domain'] for r in raw}):
candidates=[r for r in raw if r['domain']==domain and len(r['context'])<900000]
rng.shuffle(candidates)
bins=[[],[]]
for row in candidates:
prompt=prompt_for(row)
text=tokenizer.apply_chat_template([dict(role='user',content=prompt)],tokenize=False,
add_generation_prompt=True,reasoning_effort='high')
count=len(tokenizer(text,add_special_tokens=False).input_ids)
if 8192 <= count <= 245000:
bins[int(count>64000)].append(dict(id=row['_id'],domain=domain,prompt_tokens=count,
prompt_sha256=hashlib.sha256(prompt.encode()).hexdigest()))
if all(len(b)>=2 for b in bins):
break
selected=bins[0][:2]+bins[1][:2]
selected_ids={r['id'] for r in selected}
selected += [r for r in bins[0]+bins[1] if r['id'] not in selected_ids][:4-len(selected)]
assert len(selected)==4,(domain,len(selected))
chosen.extend(selected)
print(domain,[r['prompt_tokens'] for r in selected],flush=True)
result=dict(dataset='zai-org/LongBench-v2',revision=REV,source_revision=SOURCE_REV,seed=20260909,
method='Four cases per domain; two <=64K and two >64K tokens where available; 8192..245000 token limit, context <900000 characters, no truncation.',records=chosen)
(ROOT/'longbench_selection.json').write_text(json.dumps(result,indent=2))
if __name__=='__main__':main()