"""Select a reproducible, domain-balanced LongBench v2 diagnostic subset.""" import hashlib import json import random from pathlib import Path from huggingface_hub import hf_hub_download from transformers import AutoTokenizer ROOT=Path(__file__).resolve().parent REV='2b48e494f2c7a2f0af81aae178e05c7e1dde0fe9' SOURCE_REV='87420286149d9cce9bd46cd335ef9bda33c37c1b' def prompt_for(row): return ('Read the context and answer the question.\n\nContext:\n'+row['context']+ '\n\nQuestion: '+row['question']+'\n'+ '\n'.join(letter+'. '+row['choice_'+letter] for letter in 'ABCD')+ '\nPut your final answer letter inside \\boxed{}.') def main(): raw=json.loads(Path(hf_hub_download('zai-org/LongBench-v2','data.json',repo_type='dataset',revision=REV)).read_text()) tokenizer=AutoTokenizer.from_pretrained('nex-agi/Nex-N2.5-mini',revision=SOURCE_REV) rng=random.Random(20260909) chosen=[] for domain in sorted({r['domain'] for r in raw}): candidates=[r for r in raw if r['domain']==domain and len(r['context'])<900000] rng.shuffle(candidates) bins=[[],[]] for row in candidates: prompt=prompt_for(row) text=tokenizer.apply_chat_template([dict(role='user',content=prompt)],tokenize=False, add_generation_prompt=True,reasoning_effort='high') count=len(tokenizer(text,add_special_tokens=False).input_ids) if 8192 <= count <= 245000: bins[int(count>64000)].append(dict(id=row['_id'],domain=domain,prompt_tokens=count, prompt_sha256=hashlib.sha256(prompt.encode()).hexdigest())) if all(len(b)>=2 for b in bins): break selected=bins[0][:2]+bins[1][:2] selected_ids={r['id'] for r in selected} selected += [r for r in bins[0]+bins[1] if r['id'] not in selected_ids][:4-len(selected)] assert len(selected)==4,(domain,len(selected)) chosen.extend(selected) print(domain,[r['prompt_tokens'] for r in selected],flush=True) result=dict(dataset='zai-org/LongBench-v2',revision=REV,source_revision=SOURCE_REV,seed=20260909, method='Four cases per domain; two <=64K and two >64K tokens where available; 8192..245000 token limit, context <900000 characters, no truncation.',records=chosen) (ROOT/'longbench_selection.json').write_text(json.dumps(result,indent=2)) if __name__=='__main__':main()