File size: 2,487 Bytes
63f6cfd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
"""Select a reproducible, domain-balanced LongBench v2 diagnostic subset."""
import hashlib
import json
import random
from pathlib import Path
from huggingface_hub import hf_hub_download
from transformers import AutoTokenizer

ROOT=Path(__file__).resolve().parent
REV='2b48e494f2c7a2f0af81aae178e05c7e1dde0fe9'
SOURCE_REV='87420286149d9cce9bd46cd335ef9bda33c37c1b'

def prompt_for(row):
    return ('Read the context and answer the question.\n\nContext:\n'+row['context']+
            '\n\nQuestion: '+row['question']+'\n'+
            '\n'.join(letter+'. '+row['choice_'+letter] for letter in 'ABCD')+
            '\nPut your final answer letter inside \\boxed{}.')

def main():
    raw=json.loads(Path(hf_hub_download('zai-org/LongBench-v2','data.json',repo_type='dataset',revision=REV)).read_text())
    tokenizer=AutoTokenizer.from_pretrained('nex-agi/Nex-N2.5-mini',revision=SOURCE_REV)
    rng=random.Random(20260909)
    chosen=[]
    for domain in sorted({r['domain'] for r in raw}):
        candidates=[r for r in raw if r['domain']==domain and len(r['context'])<900000]
        rng.shuffle(candidates)
        bins=[[],[]]
        for row in candidates:
            prompt=prompt_for(row)
            text=tokenizer.apply_chat_template([dict(role='user',content=prompt)],tokenize=False,
                                              add_generation_prompt=True,reasoning_effort='high')
            count=len(tokenizer(text,add_special_tokens=False).input_ids)
            if 8192 <= count <= 245000:
                bins[int(count>64000)].append(dict(id=row['_id'],domain=domain,prompt_tokens=count,
                    prompt_sha256=hashlib.sha256(prompt.encode()).hexdigest()))
            if all(len(b)>=2 for b in bins):
                break
        selected=bins[0][:2]+bins[1][:2]
        selected_ids={r['id'] for r in selected}
        selected += [r for r in bins[0]+bins[1] if r['id'] not in selected_ids][:4-len(selected)]
        assert len(selected)==4,(domain,len(selected))
        chosen.extend(selected)
        print(domain,[r['prompt_tokens'] for r in selected],flush=True)
    result=dict(dataset='zai-org/LongBench-v2',revision=REV,source_revision=SOURCE_REV,seed=20260909,
        method='Four cases per domain; two <=64K and two >64K tokens where available; 8192..245000 token limit, context <900000 characters, no truncation.',records=chosen)
    (ROOT/'longbench_selection.json').write_text(json.dumps(result,indent=2))

if __name__=='__main__':main()