chaoliangUNSW commited on
Commit
11ce5d7
·
verified ·
1 Parent(s): 1ea4cb7

Jev-Style-2B-Decision-v3: release

Browse files
Files changed (47) hide show
  1. .gitattributes +5 -0
  2. 8bit/chat_template.jinja +154 -0
  3. 8bit/config.json +85 -0
  4. 8bit/generation_config.json +6 -0
  5. 8bit/macjev_norms_fp32.safetensors +3 -0
  6. 8bit/model.safetensors +3 -0
  7. 8bit/model.safetensors.index.json +702 -0
  8. 8bit/tokenizer.json +3 -0
  9. 8bit/tokenizer_config.json +33 -0
  10. LICENSE +202 -0
  11. NOTICE +35 -0
  12. README.md +194 -0
  13. THIRD_PARTY_NOTICES.md +41 -0
  14. bf16/chat_template.jinja +154 -0
  15. bf16/config.json +75 -0
  16. bf16/generation_config.json +6 -0
  17. bf16/macjev_norms_fp32.safetensors +3 -0
  18. bf16/model.safetensors +3 -0
  19. bf16/model.safetensors.index.json +328 -0
  20. bf16/tokenizer.json +3 -0
  21. bf16/tokenizer_config.json +33 -0
  22. config.json +75 -0
  23. figures/banner.data.json +44 -0
  24. figures/banner.png +3 -0
  25. figures/jevbench.data.json +49 -0
  26. figures/jevbench.png +3 -0
  27. figures/jevbench.svg +245 -0
  28. figures/zeroshot.data.json +40 -0
  29. figures/zeroshot.png +3 -0
  30. figures/zeroshot.svg +253 -0
  31. jev_style_decision_mlx.py +1059 -0
  32. manifest.json +190 -0
  33. readout_config.json +47 -0
  34. release_config.json +422 -0
  35. requirements.txt +8 -0
  36. validation/SOURCES.json +60 -0
  37. validation/latency_2b.json +1568 -0
  38. validation/parity/PREDECLARED_RELEASE_GATES_2B.md +29 -0
  39. validation/parity/compare_mlx_affine8-g64_base_vs_block.json +276 -0
  40. validation/parity/compare_mlx_affine8-g64_gate_vs_block.json +228 -0
  41. validation/parity/compare_mlx_bf16_base_vs_block.json +269 -0
  42. validation/parity/compare_mlx_bf16_gate_vs_block.json +221 -0
  43. validation/parity/cross_format_dp.json +68 -0
  44. validation/runtime/fixture_verify_8bit.json +319 -0
  45. validation/runtime/fixture_verify_bf16.json +319 -0
  46. validation/runtime/render_verify.json +88 -0
  47. validation/runtime/reverify_mlx.json +23 -0
.gitattributes CHANGED
@@ -33,3 +33,8 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ 8bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ bf16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ figures/banner.png filter=lfs diff=lfs merge=lfs -text
39
+ figures/jevbench.png filter=lfs diff=lfs merge=lfs -text
40
+ figures/zeroshot.png filter=lfs diff=lfs merge=lfs -text
8bit/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is true %}
150
+ {{- '<think>\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n\n</think>\n\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
8bit/config.json ADDED
@@ -0,0 +1,85 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 2048,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 6144,
17
+ "layer_types": [
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "linear_attention",
21
+ "full_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention"
42
+ ],
43
+ "linear_conv_kernel_dim": 4,
44
+ "linear_key_head_dim": 128,
45
+ "linear_num_key_heads": 16,
46
+ "linear_num_value_heads": 16,
47
+ "linear_value_head_dim": 128,
48
+ "mamba_ssm_dtype": "float32",
49
+ "max_position_embeddings": 262144,
50
+ "mlp_only_layers": [],
51
+ "model_type": "qwen3_5",
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_hidden_layers": 24,
56
+ "num_key_value_heads": 2,
57
+ "pad_token_id": null,
58
+ "partial_rotary_factor": 0.25,
59
+ "quantization": {
60
+ "group_size": 64,
61
+ "bits": 8,
62
+ "mode": "affine"
63
+ },
64
+ "quantization_config": {
65
+ "group_size": 64,
66
+ "bits": 8,
67
+ "mode": "affine"
68
+ },
69
+ "rms_norm_eps": 1e-06,
70
+ "rope_parameters": {
71
+ "mrope_interleaved": true,
72
+ "mrope_section": [
73
+ 11,
74
+ 11,
75
+ 10
76
+ ],
77
+ "partial_rotary_factor": 0.25,
78
+ "rope_theta": 10000000,
79
+ "type": "default"
80
+ },
81
+ "tie_word_embeddings": true,
82
+ "transformers_version": "5.17.0",
83
+ "use_cache": false,
84
+ "vocab_size": 248320
85
+ }
8bit/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "eos_token_id": 248044,
4
+ "transformers_version": "5.17.0",
5
+ "use_cache": true
6
+ }
8bit/macjev_norms_fp32.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:02f08c422228929c44dd84856678ea95d41401138aa56b1b771a654376d3e51c
3
+ size 420112
8bit/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:184b4dda2f1285fe95a00a6271b364c861d370fbf913030f0259b2423f88a740
3
+ size 2000043057
8bit/model.safetensors.index.json ADDED
@@ -0,0 +1,702 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 1999953536,
4
+ "total_parameters": 1881824512
5
+ },
6
+ "weight_map": {
7
+ "language_model.model.embed_tokens.biases": "model.safetensors",
8
+ "language_model.model.embed_tokens.scales": "model.safetensors",
9
+ "language_model.model.embed_tokens.weight": "model.safetensors",
10
+ "language_model.model.layers.0.input_layernorm.weight": "model.safetensors",
11
+ "language_model.model.layers.0.linear_attn.A_log": "model.safetensors",
12
+ "language_model.model.layers.0.linear_attn.conv1d.weight": "model.safetensors",
13
+ "language_model.model.layers.0.linear_attn.dt_bias": "model.safetensors",
14
+ "language_model.model.layers.0.linear_attn.in_proj_a.biases": "model.safetensors",
15
+ "language_model.model.layers.0.linear_attn.in_proj_a.scales": "model.safetensors",
16
+ "language_model.model.layers.0.linear_attn.in_proj_a.weight": "model.safetensors",
17
+ "language_model.model.layers.0.linear_attn.in_proj_b.biases": "model.safetensors",
18
+ "language_model.model.layers.0.linear_attn.in_proj_b.scales": "model.safetensors",
19
+ "language_model.model.layers.0.linear_attn.in_proj_b.weight": "model.safetensors",
20
+ "language_model.model.layers.0.linear_attn.in_proj_qkv.biases": "model.safetensors",
21
+ "language_model.model.layers.0.linear_attn.in_proj_qkv.scales": "model.safetensors",
22
+ "language_model.model.layers.0.linear_attn.in_proj_qkv.weight": "model.safetensors",
23
+ "language_model.model.layers.0.linear_attn.in_proj_z.biases": "model.safetensors",
24
+ "language_model.model.layers.0.linear_attn.in_proj_z.scales": "model.safetensors",
25
+ "language_model.model.layers.0.linear_attn.in_proj_z.weight": "model.safetensors",
26
+ "language_model.model.layers.0.linear_attn.norm.weight": "model.safetensors",
27
+ "language_model.model.layers.0.linear_attn.out_proj.biases": "model.safetensors",
28
+ "language_model.model.layers.0.linear_attn.out_proj.scales": "model.safetensors",
29
+ "language_model.model.layers.0.linear_attn.out_proj.weight": "model.safetensors",
30
+ "language_model.model.layers.0.mlp.down_proj.biases": "model.safetensors",
31
+ "language_model.model.layers.0.mlp.down_proj.scales": "model.safetensors",
32
+ "language_model.model.layers.0.mlp.down_proj.weight": "model.safetensors",
33
+ "language_model.model.layers.0.mlp.gate_proj.biases": "model.safetensors",
34
+ "language_model.model.layers.0.mlp.gate_proj.scales": "model.safetensors",
35
+ "language_model.model.layers.0.mlp.gate_proj.weight": "model.safetensors",
36
+ "language_model.model.layers.0.mlp.up_proj.biases": "model.safetensors",
37
+ "language_model.model.layers.0.mlp.up_proj.scales": "model.safetensors",
38
+ "language_model.model.layers.0.mlp.up_proj.weight": "model.safetensors",
39
+ "language_model.model.layers.0.post_attention_layernorm.weight": "model.safetensors",
40
+ "language_model.model.layers.1.input_layernorm.weight": "model.safetensors",
41
+ "language_model.model.layers.1.linear_attn.A_log": "model.safetensors",
42
+ "language_model.model.layers.1.linear_attn.conv1d.weight": "model.safetensors",
43
+ "language_model.model.layers.1.linear_attn.dt_bias": "model.safetensors",
44
+ "language_model.model.layers.1.linear_attn.in_proj_a.biases": "model.safetensors",
45
+ "language_model.model.layers.1.linear_attn.in_proj_a.scales": "model.safetensors",
46
+ "language_model.model.layers.1.linear_attn.in_proj_a.weight": "model.safetensors",
47
+ "language_model.model.layers.1.linear_attn.in_proj_b.biases": "model.safetensors",
48
+ "language_model.model.layers.1.linear_attn.in_proj_b.scales": "model.safetensors",
49
+ "language_model.model.layers.1.linear_attn.in_proj_b.weight": "model.safetensors",
50
+ "language_model.model.layers.1.linear_attn.in_proj_qkv.biases": "model.safetensors",
51
+ "language_model.model.layers.1.linear_attn.in_proj_qkv.scales": "model.safetensors",
52
+ "language_model.model.layers.1.linear_attn.in_proj_qkv.weight": "model.safetensors",
53
+ "language_model.model.layers.1.linear_attn.in_proj_z.biases": "model.safetensors",
54
+ "language_model.model.layers.1.linear_attn.in_proj_z.scales": "model.safetensors",
55
+ "language_model.model.layers.1.linear_attn.in_proj_z.weight": "model.safetensors",
56
+ "language_model.model.layers.1.linear_attn.norm.weight": "model.safetensors",
57
+ "language_model.model.layers.1.linear_attn.out_proj.biases": "model.safetensors",
58
+ "language_model.model.layers.1.linear_attn.out_proj.scales": "model.safetensors",
59
+ "language_model.model.layers.1.linear_attn.out_proj.weight": "model.safetensors",
60
+ "language_model.model.layers.1.mlp.down_proj.biases": "model.safetensors",
61
+ "language_model.model.layers.1.mlp.down_proj.scales": "model.safetensors",
62
+ "language_model.model.layers.1.mlp.down_proj.weight": "model.safetensors",
63
+ "language_model.model.layers.1.mlp.gate_proj.biases": "model.safetensors",
64
+ "language_model.model.layers.1.mlp.gate_proj.scales": "model.safetensors",
65
+ "language_model.model.layers.1.mlp.gate_proj.weight": "model.safetensors",
66
+ "language_model.model.layers.1.mlp.up_proj.biases": "model.safetensors",
67
+ "language_model.model.layers.1.mlp.up_proj.scales": "model.safetensors",
68
+ "language_model.model.layers.1.mlp.up_proj.weight": "model.safetensors",
69
+ "language_model.model.layers.1.post_attention_layernorm.weight": "model.safetensors",
70
+ "language_model.model.layers.10.input_layernorm.weight": "model.safetensors",
71
+ "language_model.model.layers.10.linear_attn.A_log": "model.safetensors",
72
+ "language_model.model.layers.10.linear_attn.conv1d.weight": "model.safetensors",
73
+ "language_model.model.layers.10.linear_attn.dt_bias": "model.safetensors",
74
+ "language_model.model.layers.10.linear_attn.in_proj_a.biases": "model.safetensors",
75
+ "language_model.model.layers.10.linear_attn.in_proj_a.scales": "model.safetensors",
76
+ "language_model.model.layers.10.linear_attn.in_proj_a.weight": "model.safetensors",
77
+ "language_model.model.layers.10.linear_attn.in_proj_b.biases": "model.safetensors",
78
+ "language_model.model.layers.10.linear_attn.in_proj_b.scales": "model.safetensors",
79
+ "language_model.model.layers.10.linear_attn.in_proj_b.weight": "model.safetensors",
80
+ "language_model.model.layers.10.linear_attn.in_proj_qkv.biases": "model.safetensors",
81
+ "language_model.model.layers.10.linear_attn.in_proj_qkv.scales": "model.safetensors",
82
+ "language_model.model.layers.10.linear_attn.in_proj_qkv.weight": "model.safetensors",
83
+ "language_model.model.layers.10.linear_attn.in_proj_z.biases": "model.safetensors",
84
+ "language_model.model.layers.10.linear_attn.in_proj_z.scales": "model.safetensors",
85
+ "language_model.model.layers.10.linear_attn.in_proj_z.weight": "model.safetensors",
86
+ "language_model.model.layers.10.linear_attn.norm.weight": "model.safetensors",
87
+ "language_model.model.layers.10.linear_attn.out_proj.biases": "model.safetensors",
88
+ "language_model.model.layers.10.linear_attn.out_proj.scales": "model.safetensors",
89
+ "language_model.model.layers.10.linear_attn.out_proj.weight": "model.safetensors",
90
+ "language_model.model.layers.10.mlp.down_proj.biases": "model.safetensors",
91
+ "language_model.model.layers.10.mlp.down_proj.scales": "model.safetensors",
92
+ "language_model.model.layers.10.mlp.down_proj.weight": "model.safetensors",
93
+ "language_model.model.layers.10.mlp.gate_proj.biases": "model.safetensors",
94
+ "language_model.model.layers.10.mlp.gate_proj.scales": "model.safetensors",
95
+ "language_model.model.layers.10.mlp.gate_proj.weight": "model.safetensors",
96
+ "language_model.model.layers.10.mlp.up_proj.biases": "model.safetensors",
97
+ "language_model.model.layers.10.mlp.up_proj.scales": "model.safetensors",
98
+ "language_model.model.layers.10.mlp.up_proj.weight": "model.safetensors",
99
+ "language_model.model.layers.10.post_attention_layernorm.weight": "model.safetensors",
100
+ "language_model.model.layers.11.input_layernorm.weight": "model.safetensors",
101
+ "language_model.model.layers.11.mlp.down_proj.biases": "model.safetensors",
102
+ "language_model.model.layers.11.mlp.down_proj.scales": "model.safetensors",
103
+ "language_model.model.layers.11.mlp.down_proj.weight": "model.safetensors",
104
+ "language_model.model.layers.11.mlp.gate_proj.biases": "model.safetensors",
105
+ "language_model.model.layers.11.mlp.gate_proj.scales": "model.safetensors",
106
+ "language_model.model.layers.11.mlp.gate_proj.weight": "model.safetensors",
107
+ "language_model.model.layers.11.mlp.up_proj.biases": "model.safetensors",
108
+ "language_model.model.layers.11.mlp.up_proj.scales": "model.safetensors",
109
+ "language_model.model.layers.11.mlp.up_proj.weight": "model.safetensors",
110
+ "language_model.model.layers.11.post_attention_layernorm.weight": "model.safetensors",
111
+ "language_model.model.layers.11.self_attn.k_norm.weight": "model.safetensors",
112
+ "language_model.model.layers.11.self_attn.k_proj.biases": "model.safetensors",
113
+ "language_model.model.layers.11.self_attn.k_proj.scales": "model.safetensors",
114
+ "language_model.model.layers.11.self_attn.k_proj.weight": "model.safetensors",
115
+ "language_model.model.layers.11.self_attn.o_proj.biases": "model.safetensors",
116
+ "language_model.model.layers.11.self_attn.o_proj.scales": "model.safetensors",
117
+ "language_model.model.layers.11.self_attn.o_proj.weight": "model.safetensors",
118
+ "language_model.model.layers.11.self_attn.q_norm.weight": "model.safetensors",
119
+ "language_model.model.layers.11.self_attn.q_proj.biases": "model.safetensors",
120
+ "language_model.model.layers.11.self_attn.q_proj.scales": "model.safetensors",
121
+ "language_model.model.layers.11.self_attn.q_proj.weight": "model.safetensors",
122
+ "language_model.model.layers.11.self_attn.v_proj.biases": "model.safetensors",
123
+ "language_model.model.layers.11.self_attn.v_proj.scales": "model.safetensors",
124
+ "language_model.model.layers.11.self_attn.v_proj.weight": "model.safetensors",
125
+ "language_model.model.layers.12.input_layernorm.weight": "model.safetensors",
126
+ "language_model.model.layers.12.linear_attn.A_log": "model.safetensors",
127
+ "language_model.model.layers.12.linear_attn.conv1d.weight": "model.safetensors",
128
+ "language_model.model.layers.12.linear_attn.dt_bias": "model.safetensors",
129
+ "language_model.model.layers.12.linear_attn.in_proj_a.biases": "model.safetensors",
130
+ "language_model.model.layers.12.linear_attn.in_proj_a.scales": "model.safetensors",
131
+ "language_model.model.layers.12.linear_attn.in_proj_a.weight": "model.safetensors",
132
+ "language_model.model.layers.12.linear_attn.in_proj_b.biases": "model.safetensors",
133
+ "language_model.model.layers.12.linear_attn.in_proj_b.scales": "model.safetensors",
134
+ "language_model.model.layers.12.linear_attn.in_proj_b.weight": "model.safetensors",
135
+ "language_model.model.layers.12.linear_attn.in_proj_qkv.biases": "model.safetensors",
136
+ "language_model.model.layers.12.linear_attn.in_proj_qkv.scales": "model.safetensors",
137
+ "language_model.model.layers.12.linear_attn.in_proj_qkv.weight": "model.safetensors",
138
+ "language_model.model.layers.12.linear_attn.in_proj_z.biases": "model.safetensors",
139
+ "language_model.model.layers.12.linear_attn.in_proj_z.scales": "model.safetensors",
140
+ "language_model.model.layers.12.linear_attn.in_proj_z.weight": "model.safetensors",
141
+ "language_model.model.layers.12.linear_attn.norm.weight": "model.safetensors",
142
+ "language_model.model.layers.12.linear_attn.out_proj.biases": "model.safetensors",
143
+ "language_model.model.layers.12.linear_attn.out_proj.scales": "model.safetensors",
144
+ "language_model.model.layers.12.linear_attn.out_proj.weight": "model.safetensors",
145
+ "language_model.model.layers.12.mlp.down_proj.biases": "model.safetensors",
146
+ "language_model.model.layers.12.mlp.down_proj.scales": "model.safetensors",
147
+ "language_model.model.layers.12.mlp.down_proj.weight": "model.safetensors",
148
+ "language_model.model.layers.12.mlp.gate_proj.biases": "model.safetensors",
149
+ "language_model.model.layers.12.mlp.gate_proj.scales": "model.safetensors",
150
+ "language_model.model.layers.12.mlp.gate_proj.weight": "model.safetensors",
151
+ "language_model.model.layers.12.mlp.up_proj.biases": "model.safetensors",
152
+ "language_model.model.layers.12.mlp.up_proj.scales": "model.safetensors",
153
+ "language_model.model.layers.12.mlp.up_proj.weight": "model.safetensors",
154
+ "language_model.model.layers.12.post_attention_layernorm.weight": "model.safetensors",
155
+ "language_model.model.layers.13.input_layernorm.weight": "model.safetensors",
156
+ "language_model.model.layers.13.linear_attn.A_log": "model.safetensors",
157
+ "language_model.model.layers.13.linear_attn.conv1d.weight": "model.safetensors",
158
+ "language_model.model.layers.13.linear_attn.dt_bias": "model.safetensors",
159
+ "language_model.model.layers.13.linear_attn.in_proj_a.biases": "model.safetensors",
160
+ "language_model.model.layers.13.linear_attn.in_proj_a.scales": "model.safetensors",
161
+ "language_model.model.layers.13.linear_attn.in_proj_a.weight": "model.safetensors",
162
+ "language_model.model.layers.13.linear_attn.in_proj_b.biases": "model.safetensors",
163
+ "language_model.model.layers.13.linear_attn.in_proj_b.scales": "model.safetensors",
164
+ "language_model.model.layers.13.linear_attn.in_proj_b.weight": "model.safetensors",
165
+ "language_model.model.layers.13.linear_attn.in_proj_qkv.biases": "model.safetensors",
166
+ "language_model.model.layers.13.linear_attn.in_proj_qkv.scales": "model.safetensors",
167
+ "language_model.model.layers.13.linear_attn.in_proj_qkv.weight": "model.safetensors",
168
+ "language_model.model.layers.13.linear_attn.in_proj_z.biases": "model.safetensors",
169
+ "language_model.model.layers.13.linear_attn.in_proj_z.scales": "model.safetensors",
170
+ "language_model.model.layers.13.linear_attn.in_proj_z.weight": "model.safetensors",
171
+ "language_model.model.layers.13.linear_attn.norm.weight": "model.safetensors",
172
+ "language_model.model.layers.13.linear_attn.out_proj.biases": "model.safetensors",
173
+ "language_model.model.layers.13.linear_attn.out_proj.scales": "model.safetensors",
174
+ "language_model.model.layers.13.linear_attn.out_proj.weight": "model.safetensors",
175
+ "language_model.model.layers.13.mlp.down_proj.biases": "model.safetensors",
176
+ "language_model.model.layers.13.mlp.down_proj.scales": "model.safetensors",
177
+ "language_model.model.layers.13.mlp.down_proj.weight": "model.safetensors",
178
+ "language_model.model.layers.13.mlp.gate_proj.biases": "model.safetensors",
179
+ "language_model.model.layers.13.mlp.gate_proj.scales": "model.safetensors",
180
+ "language_model.model.layers.13.mlp.gate_proj.weight": "model.safetensors",
181
+ "language_model.model.layers.13.mlp.up_proj.biases": "model.safetensors",
182
+ "language_model.model.layers.13.mlp.up_proj.scales": "model.safetensors",
183
+ "language_model.model.layers.13.mlp.up_proj.weight": "model.safetensors",
184
+ "language_model.model.layers.13.post_attention_layernorm.weight": "model.safetensors",
185
+ "language_model.model.layers.14.input_layernorm.weight": "model.safetensors",
186
+ "language_model.model.layers.14.linear_attn.A_log": "model.safetensors",
187
+ "language_model.model.layers.14.linear_attn.conv1d.weight": "model.safetensors",
188
+ "language_model.model.layers.14.linear_attn.dt_bias": "model.safetensors",
189
+ "language_model.model.layers.14.linear_attn.in_proj_a.biases": "model.safetensors",
190
+ "language_model.model.layers.14.linear_attn.in_proj_a.scales": "model.safetensors",
191
+ "language_model.model.layers.14.linear_attn.in_proj_a.weight": "model.safetensors",
192
+ "language_model.model.layers.14.linear_attn.in_proj_b.biases": "model.safetensors",
193
+ "language_model.model.layers.14.linear_attn.in_proj_b.scales": "model.safetensors",
194
+ "language_model.model.layers.14.linear_attn.in_proj_b.weight": "model.safetensors",
195
+ "language_model.model.layers.14.linear_attn.in_proj_qkv.biases": "model.safetensors",
196
+ "language_model.model.layers.14.linear_attn.in_proj_qkv.scales": "model.safetensors",
197
+ "language_model.model.layers.14.linear_attn.in_proj_qkv.weight": "model.safetensors",
198
+ "language_model.model.layers.14.linear_attn.in_proj_z.biases": "model.safetensors",
199
+ "language_model.model.layers.14.linear_attn.in_proj_z.scales": "model.safetensors",
200
+ "language_model.model.layers.14.linear_attn.in_proj_z.weight": "model.safetensors",
201
+ "language_model.model.layers.14.linear_attn.norm.weight": "model.safetensors",
202
+ "language_model.model.layers.14.linear_attn.out_proj.biases": "model.safetensors",
203
+ "language_model.model.layers.14.linear_attn.out_proj.scales": "model.safetensors",
204
+ "language_model.model.layers.14.linear_attn.out_proj.weight": "model.safetensors",
205
+ "language_model.model.layers.14.mlp.down_proj.biases": "model.safetensors",
206
+ "language_model.model.layers.14.mlp.down_proj.scales": "model.safetensors",
207
+ "language_model.model.layers.14.mlp.down_proj.weight": "model.safetensors",
208
+ "language_model.model.layers.14.mlp.gate_proj.biases": "model.safetensors",
209
+ "language_model.model.layers.14.mlp.gate_proj.scales": "model.safetensors",
210
+ "language_model.model.layers.14.mlp.gate_proj.weight": "model.safetensors",
211
+ "language_model.model.layers.14.mlp.up_proj.biases": "model.safetensors",
212
+ "language_model.model.layers.14.mlp.up_proj.scales": "model.safetensors",
213
+ "language_model.model.layers.14.mlp.up_proj.weight": "model.safetensors",
214
+ "language_model.model.layers.14.post_attention_layernorm.weight": "model.safetensors",
215
+ "language_model.model.layers.15.input_layernorm.weight": "model.safetensors",
216
+ "language_model.model.layers.15.mlp.down_proj.biases": "model.safetensors",
217
+ "language_model.model.layers.15.mlp.down_proj.scales": "model.safetensors",
218
+ "language_model.model.layers.15.mlp.down_proj.weight": "model.safetensors",
219
+ "language_model.model.layers.15.mlp.gate_proj.biases": "model.safetensors",
220
+ "language_model.model.layers.15.mlp.gate_proj.scales": "model.safetensors",
221
+ "language_model.model.layers.15.mlp.gate_proj.weight": "model.safetensors",
222
+ "language_model.model.layers.15.mlp.up_proj.biases": "model.safetensors",
223
+ "language_model.model.layers.15.mlp.up_proj.scales": "model.safetensors",
224
+ "language_model.model.layers.15.mlp.up_proj.weight": "model.safetensors",
225
+ "language_model.model.layers.15.post_attention_layernorm.weight": "model.safetensors",
226
+ "language_model.model.layers.15.self_attn.k_norm.weight": "model.safetensors",
227
+ "language_model.model.layers.15.self_attn.k_proj.biases": "model.safetensors",
228
+ "language_model.model.layers.15.self_attn.k_proj.scales": "model.safetensors",
229
+ "language_model.model.layers.15.self_attn.k_proj.weight": "model.safetensors",
230
+ "language_model.model.layers.15.self_attn.o_proj.biases": "model.safetensors",
231
+ "language_model.model.layers.15.self_attn.o_proj.scales": "model.safetensors",
232
+ "language_model.model.layers.15.self_attn.o_proj.weight": "model.safetensors",
233
+ "language_model.model.layers.15.self_attn.q_norm.weight": "model.safetensors",
234
+ "language_model.model.layers.15.self_attn.q_proj.biases": "model.safetensors",
235
+ "language_model.model.layers.15.self_attn.q_proj.scales": "model.safetensors",
236
+ "language_model.model.layers.15.self_attn.q_proj.weight": "model.safetensors",
237
+ "language_model.model.layers.15.self_attn.v_proj.biases": "model.safetensors",
238
+ "language_model.model.layers.15.self_attn.v_proj.scales": "model.safetensors",
239
+ "language_model.model.layers.15.self_attn.v_proj.weight": "model.safetensors",
240
+ "language_model.model.layers.16.input_layernorm.weight": "model.safetensors",
241
+ "language_model.model.layers.16.linear_attn.A_log": "model.safetensors",
242
+ "language_model.model.layers.16.linear_attn.conv1d.weight": "model.safetensors",
243
+ "language_model.model.layers.16.linear_attn.dt_bias": "model.safetensors",
244
+ "language_model.model.layers.16.linear_attn.in_proj_a.biases": "model.safetensors",
245
+ "language_model.model.layers.16.linear_attn.in_proj_a.scales": "model.safetensors",
246
+ "language_model.model.layers.16.linear_attn.in_proj_a.weight": "model.safetensors",
247
+ "language_model.model.layers.16.linear_attn.in_proj_b.biases": "model.safetensors",
248
+ "language_model.model.layers.16.linear_attn.in_proj_b.scales": "model.safetensors",
249
+ "language_model.model.layers.16.linear_attn.in_proj_b.weight": "model.safetensors",
250
+ "language_model.model.layers.16.linear_attn.in_proj_qkv.biases": "model.safetensors",
251
+ "language_model.model.layers.16.linear_attn.in_proj_qkv.scales": "model.safetensors",
252
+ "language_model.model.layers.16.linear_attn.in_proj_qkv.weight": "model.safetensors",
253
+ "language_model.model.layers.16.linear_attn.in_proj_z.biases": "model.safetensors",
254
+ "language_model.model.layers.16.linear_attn.in_proj_z.scales": "model.safetensors",
255
+ "language_model.model.layers.16.linear_attn.in_proj_z.weight": "model.safetensors",
256
+ "language_model.model.layers.16.linear_attn.norm.weight": "model.safetensors",
257
+ "language_model.model.layers.16.linear_attn.out_proj.biases": "model.safetensors",
258
+ "language_model.model.layers.16.linear_attn.out_proj.scales": "model.safetensors",
259
+ "language_model.model.layers.16.linear_attn.out_proj.weight": "model.safetensors",
260
+ "language_model.model.layers.16.mlp.down_proj.biases": "model.safetensors",
261
+ "language_model.model.layers.16.mlp.down_proj.scales": "model.safetensors",
262
+ "language_model.model.layers.16.mlp.down_proj.weight": "model.safetensors",
263
+ "language_model.model.layers.16.mlp.gate_proj.biases": "model.safetensors",
264
+ "language_model.model.layers.16.mlp.gate_proj.scales": "model.safetensors",
265
+ "language_model.model.layers.16.mlp.gate_proj.weight": "model.safetensors",
266
+ "language_model.model.layers.16.mlp.up_proj.biases": "model.safetensors",
267
+ "language_model.model.layers.16.mlp.up_proj.scales": "model.safetensors",
268
+ "language_model.model.layers.16.mlp.up_proj.weight": "model.safetensors",
269
+ "language_model.model.layers.16.post_attention_layernorm.weight": "model.safetensors",
270
+ "language_model.model.layers.17.input_layernorm.weight": "model.safetensors",
271
+ "language_model.model.layers.17.linear_attn.A_log": "model.safetensors",
272
+ "language_model.model.layers.17.linear_attn.conv1d.weight": "model.safetensors",
273
+ "language_model.model.layers.17.linear_attn.dt_bias": "model.safetensors",
274
+ "language_model.model.layers.17.linear_attn.in_proj_a.biases": "model.safetensors",
275
+ "language_model.model.layers.17.linear_attn.in_proj_a.scales": "model.safetensors",
276
+ "language_model.model.layers.17.linear_attn.in_proj_a.weight": "model.safetensors",
277
+ "language_model.model.layers.17.linear_attn.in_proj_b.biases": "model.safetensors",
278
+ "language_model.model.layers.17.linear_attn.in_proj_b.scales": "model.safetensors",
279
+ "language_model.model.layers.17.linear_attn.in_proj_b.weight": "model.safetensors",
280
+ "language_model.model.layers.17.linear_attn.in_proj_qkv.biases": "model.safetensors",
281
+ "language_model.model.layers.17.linear_attn.in_proj_qkv.scales": "model.safetensors",
282
+ "language_model.model.layers.17.linear_attn.in_proj_qkv.weight": "model.safetensors",
283
+ "language_model.model.layers.17.linear_attn.in_proj_z.biases": "model.safetensors",
284
+ "language_model.model.layers.17.linear_attn.in_proj_z.scales": "model.safetensors",
285
+ "language_model.model.layers.17.linear_attn.in_proj_z.weight": "model.safetensors",
286
+ "language_model.model.layers.17.linear_attn.norm.weight": "model.safetensors",
287
+ "language_model.model.layers.17.linear_attn.out_proj.biases": "model.safetensors",
288
+ "language_model.model.layers.17.linear_attn.out_proj.scales": "model.safetensors",
289
+ "language_model.model.layers.17.linear_attn.out_proj.weight": "model.safetensors",
290
+ "language_model.model.layers.17.mlp.down_proj.biases": "model.safetensors",
291
+ "language_model.model.layers.17.mlp.down_proj.scales": "model.safetensors",
292
+ "language_model.model.layers.17.mlp.down_proj.weight": "model.safetensors",
293
+ "language_model.model.layers.17.mlp.gate_proj.biases": "model.safetensors",
294
+ "language_model.model.layers.17.mlp.gate_proj.scales": "model.safetensors",
295
+ "language_model.model.layers.17.mlp.gate_proj.weight": "model.safetensors",
296
+ "language_model.model.layers.17.mlp.up_proj.biases": "model.safetensors",
297
+ "language_model.model.layers.17.mlp.up_proj.scales": "model.safetensors",
298
+ "language_model.model.layers.17.mlp.up_proj.weight": "model.safetensors",
299
+ "language_model.model.layers.17.post_attention_layernorm.weight": "model.safetensors",
300
+ "language_model.model.layers.18.input_layernorm.weight": "model.safetensors",
301
+ "language_model.model.layers.18.linear_attn.A_log": "model.safetensors",
302
+ "language_model.model.layers.18.linear_attn.conv1d.weight": "model.safetensors",
303
+ "language_model.model.layers.18.linear_attn.dt_bias": "model.safetensors",
304
+ "language_model.model.layers.18.linear_attn.in_proj_a.biases": "model.safetensors",
305
+ "language_model.model.layers.18.linear_attn.in_proj_a.scales": "model.safetensors",
306
+ "language_model.model.layers.18.linear_attn.in_proj_a.weight": "model.safetensors",
307
+ "language_model.model.layers.18.linear_attn.in_proj_b.biases": "model.safetensors",
308
+ "language_model.model.layers.18.linear_attn.in_proj_b.scales": "model.safetensors",
309
+ "language_model.model.layers.18.linear_attn.in_proj_b.weight": "model.safetensors",
310
+ "language_model.model.layers.18.linear_attn.in_proj_qkv.biases": "model.safetensors",
311
+ "language_model.model.layers.18.linear_attn.in_proj_qkv.scales": "model.safetensors",
312
+ "language_model.model.layers.18.linear_attn.in_proj_qkv.weight": "model.safetensors",
313
+ "language_model.model.layers.18.linear_attn.in_proj_z.biases": "model.safetensors",
314
+ "language_model.model.layers.18.linear_attn.in_proj_z.scales": "model.safetensors",
315
+ "language_model.model.layers.18.linear_attn.in_proj_z.weight": "model.safetensors",
316
+ "language_model.model.layers.18.linear_attn.norm.weight": "model.safetensors",
317
+ "language_model.model.layers.18.linear_attn.out_proj.biases": "model.safetensors",
318
+ "language_model.model.layers.18.linear_attn.out_proj.scales": "model.safetensors",
319
+ "language_model.model.layers.18.linear_attn.out_proj.weight": "model.safetensors",
320
+ "language_model.model.layers.18.mlp.down_proj.biases": "model.safetensors",
321
+ "language_model.model.layers.18.mlp.down_proj.scales": "model.safetensors",
322
+ "language_model.model.layers.18.mlp.down_proj.weight": "model.safetensors",
323
+ "language_model.model.layers.18.mlp.gate_proj.biases": "model.safetensors",
324
+ "language_model.model.layers.18.mlp.gate_proj.scales": "model.safetensors",
325
+ "language_model.model.layers.18.mlp.gate_proj.weight": "model.safetensors",
326
+ "language_model.model.layers.18.mlp.up_proj.biases": "model.safetensors",
327
+ "language_model.model.layers.18.mlp.up_proj.scales": "model.safetensors",
328
+ "language_model.model.layers.18.mlp.up_proj.weight": "model.safetensors",
329
+ "language_model.model.layers.18.post_attention_layernorm.weight": "model.safetensors",
330
+ "language_model.model.layers.19.input_layernorm.weight": "model.safetensors",
331
+ "language_model.model.layers.19.mlp.down_proj.biases": "model.safetensors",
332
+ "language_model.model.layers.19.mlp.down_proj.scales": "model.safetensors",
333
+ "language_model.model.layers.19.mlp.down_proj.weight": "model.safetensors",
334
+ "language_model.model.layers.19.mlp.gate_proj.biases": "model.safetensors",
335
+ "language_model.model.layers.19.mlp.gate_proj.scales": "model.safetensors",
336
+ "language_model.model.layers.19.mlp.gate_proj.weight": "model.safetensors",
337
+ "language_model.model.layers.19.mlp.up_proj.biases": "model.safetensors",
338
+ "language_model.model.layers.19.mlp.up_proj.scales": "model.safetensors",
339
+ "language_model.model.layers.19.mlp.up_proj.weight": "model.safetensors",
340
+ "language_model.model.layers.19.post_attention_layernorm.weight": "model.safetensors",
341
+ "language_model.model.layers.19.self_attn.k_norm.weight": "model.safetensors",
342
+ "language_model.model.layers.19.self_attn.k_proj.biases": "model.safetensors",
343
+ "language_model.model.layers.19.self_attn.k_proj.scales": "model.safetensors",
344
+ "language_model.model.layers.19.self_attn.k_proj.weight": "model.safetensors",
345
+ "language_model.model.layers.19.self_attn.o_proj.biases": "model.safetensors",
346
+ "language_model.model.layers.19.self_attn.o_proj.scales": "model.safetensors",
347
+ "language_model.model.layers.19.self_attn.o_proj.weight": "model.safetensors",
348
+ "language_model.model.layers.19.self_attn.q_norm.weight": "model.safetensors",
349
+ "language_model.model.layers.19.self_attn.q_proj.biases": "model.safetensors",
350
+ "language_model.model.layers.19.self_attn.q_proj.scales": "model.safetensors",
351
+ "language_model.model.layers.19.self_attn.q_proj.weight": "model.safetensors",
352
+ "language_model.model.layers.19.self_attn.v_proj.biases": "model.safetensors",
353
+ "language_model.model.layers.19.self_attn.v_proj.scales": "model.safetensors",
354
+ "language_model.model.layers.19.self_attn.v_proj.weight": "model.safetensors",
355
+ "language_model.model.layers.2.input_layernorm.weight": "model.safetensors",
356
+ "language_model.model.layers.2.linear_attn.A_log": "model.safetensors",
357
+ "language_model.model.layers.2.linear_attn.conv1d.weight": "model.safetensors",
358
+ "language_model.model.layers.2.linear_attn.dt_bias": "model.safetensors",
359
+ "language_model.model.layers.2.linear_attn.in_proj_a.biases": "model.safetensors",
360
+ "language_model.model.layers.2.linear_attn.in_proj_a.scales": "model.safetensors",
361
+ "language_model.model.layers.2.linear_attn.in_proj_a.weight": "model.safetensors",
362
+ "language_model.model.layers.2.linear_attn.in_proj_b.biases": "model.safetensors",
363
+ "language_model.model.layers.2.linear_attn.in_proj_b.scales": "model.safetensors",
364
+ "language_model.model.layers.2.linear_attn.in_proj_b.weight": "model.safetensors",
365
+ "language_model.model.layers.2.linear_attn.in_proj_qkv.biases": "model.safetensors",
366
+ "language_model.model.layers.2.linear_attn.in_proj_qkv.scales": "model.safetensors",
367
+ "language_model.model.layers.2.linear_attn.in_proj_qkv.weight": "model.safetensors",
368
+ "language_model.model.layers.2.linear_attn.in_proj_z.biases": "model.safetensors",
369
+ "language_model.model.layers.2.linear_attn.in_proj_z.scales": "model.safetensors",
370
+ "language_model.model.layers.2.linear_attn.in_proj_z.weight": "model.safetensors",
371
+ "language_model.model.layers.2.linear_attn.norm.weight": "model.safetensors",
372
+ "language_model.model.layers.2.linear_attn.out_proj.biases": "model.safetensors",
373
+ "language_model.model.layers.2.linear_attn.out_proj.scales": "model.safetensors",
374
+ "language_model.model.layers.2.linear_attn.out_proj.weight": "model.safetensors",
375
+ "language_model.model.layers.2.mlp.down_proj.biases": "model.safetensors",
376
+ "language_model.model.layers.2.mlp.down_proj.scales": "model.safetensors",
377
+ "language_model.model.layers.2.mlp.down_proj.weight": "model.safetensors",
378
+ "language_model.model.layers.2.mlp.gate_proj.biases": "model.safetensors",
379
+ "language_model.model.layers.2.mlp.gate_proj.scales": "model.safetensors",
380
+ "language_model.model.layers.2.mlp.gate_proj.weight": "model.safetensors",
381
+ "language_model.model.layers.2.mlp.up_proj.biases": "model.safetensors",
382
+ "language_model.model.layers.2.mlp.up_proj.scales": "model.safetensors",
383
+ "language_model.model.layers.2.mlp.up_proj.weight": "model.safetensors",
384
+ "language_model.model.layers.2.post_attention_layernorm.weight": "model.safetensors",
385
+ "language_model.model.layers.20.input_layernorm.weight": "model.safetensors",
386
+ "language_model.model.layers.20.linear_attn.A_log": "model.safetensors",
387
+ "language_model.model.layers.20.linear_attn.conv1d.weight": "model.safetensors",
388
+ "language_model.model.layers.20.linear_attn.dt_bias": "model.safetensors",
389
+ "language_model.model.layers.20.linear_attn.in_proj_a.biases": "model.safetensors",
390
+ "language_model.model.layers.20.linear_attn.in_proj_a.scales": "model.safetensors",
391
+ "language_model.model.layers.20.linear_attn.in_proj_a.weight": "model.safetensors",
392
+ "language_model.model.layers.20.linear_attn.in_proj_b.biases": "model.safetensors",
393
+ "language_model.model.layers.20.linear_attn.in_proj_b.scales": "model.safetensors",
394
+ "language_model.model.layers.20.linear_attn.in_proj_b.weight": "model.safetensors",
395
+ "language_model.model.layers.20.linear_attn.in_proj_qkv.biases": "model.safetensors",
396
+ "language_model.model.layers.20.linear_attn.in_proj_qkv.scales": "model.safetensors",
397
+ "language_model.model.layers.20.linear_attn.in_proj_qkv.weight": "model.safetensors",
398
+ "language_model.model.layers.20.linear_attn.in_proj_z.biases": "model.safetensors",
399
+ "language_model.model.layers.20.linear_attn.in_proj_z.scales": "model.safetensors",
400
+ "language_model.model.layers.20.linear_attn.in_proj_z.weight": "model.safetensors",
401
+ "language_model.model.layers.20.linear_attn.norm.weight": "model.safetensors",
402
+ "language_model.model.layers.20.linear_attn.out_proj.biases": "model.safetensors",
403
+ "language_model.model.layers.20.linear_attn.out_proj.scales": "model.safetensors",
404
+ "language_model.model.layers.20.linear_attn.out_proj.weight": "model.safetensors",
405
+ "language_model.model.layers.20.mlp.down_proj.biases": "model.safetensors",
406
+ "language_model.model.layers.20.mlp.down_proj.scales": "model.safetensors",
407
+ "language_model.model.layers.20.mlp.down_proj.weight": "model.safetensors",
408
+ "language_model.model.layers.20.mlp.gate_proj.biases": "model.safetensors",
409
+ "language_model.model.layers.20.mlp.gate_proj.scales": "model.safetensors",
410
+ "language_model.model.layers.20.mlp.gate_proj.weight": "model.safetensors",
411
+ "language_model.model.layers.20.mlp.up_proj.biases": "model.safetensors",
412
+ "language_model.model.layers.20.mlp.up_proj.scales": "model.safetensors",
413
+ "language_model.model.layers.20.mlp.up_proj.weight": "model.safetensors",
414
+ "language_model.model.layers.20.post_attention_layernorm.weight": "model.safetensors",
415
+ "language_model.model.layers.21.input_layernorm.weight": "model.safetensors",
416
+ "language_model.model.layers.21.linear_attn.A_log": "model.safetensors",
417
+ "language_model.model.layers.21.linear_attn.conv1d.weight": "model.safetensors",
418
+ "language_model.model.layers.21.linear_attn.dt_bias": "model.safetensors",
419
+ "language_model.model.layers.21.linear_attn.in_proj_a.biases": "model.safetensors",
420
+ "language_model.model.layers.21.linear_attn.in_proj_a.scales": "model.safetensors",
421
+ "language_model.model.layers.21.linear_attn.in_proj_a.weight": "model.safetensors",
422
+ "language_model.model.layers.21.linear_attn.in_proj_b.biases": "model.safetensors",
423
+ "language_model.model.layers.21.linear_attn.in_proj_b.scales": "model.safetensors",
424
+ "language_model.model.layers.21.linear_attn.in_proj_b.weight": "model.safetensors",
425
+ "language_model.model.layers.21.linear_attn.in_proj_qkv.biases": "model.safetensors",
426
+ "language_model.model.layers.21.linear_attn.in_proj_qkv.scales": "model.safetensors",
427
+ "language_model.model.layers.21.linear_attn.in_proj_qkv.weight": "model.safetensors",
428
+ "language_model.model.layers.21.linear_attn.in_proj_z.biases": "model.safetensors",
429
+ "language_model.model.layers.21.linear_attn.in_proj_z.scales": "model.safetensors",
430
+ "language_model.model.layers.21.linear_attn.in_proj_z.weight": "model.safetensors",
431
+ "language_model.model.layers.21.linear_attn.norm.weight": "model.safetensors",
432
+ "language_model.model.layers.21.linear_attn.out_proj.biases": "model.safetensors",
433
+ "language_model.model.layers.21.linear_attn.out_proj.scales": "model.safetensors",
434
+ "language_model.model.layers.21.linear_attn.out_proj.weight": "model.safetensors",
435
+ "language_model.model.layers.21.mlp.down_proj.biases": "model.safetensors",
436
+ "language_model.model.layers.21.mlp.down_proj.scales": "model.safetensors",
437
+ "language_model.model.layers.21.mlp.down_proj.weight": "model.safetensors",
438
+ "language_model.model.layers.21.mlp.gate_proj.biases": "model.safetensors",
439
+ "language_model.model.layers.21.mlp.gate_proj.scales": "model.safetensors",
440
+ "language_model.model.layers.21.mlp.gate_proj.weight": "model.safetensors",
441
+ "language_model.model.layers.21.mlp.up_proj.biases": "model.safetensors",
442
+ "language_model.model.layers.21.mlp.up_proj.scales": "model.safetensors",
443
+ "language_model.model.layers.21.mlp.up_proj.weight": "model.safetensors",
444
+ "language_model.model.layers.21.post_attention_layernorm.weight": "model.safetensors",
445
+ "language_model.model.layers.22.input_layernorm.weight": "model.safetensors",
446
+ "language_model.model.layers.22.linear_attn.A_log": "model.safetensors",
447
+ "language_model.model.layers.22.linear_attn.conv1d.weight": "model.safetensors",
448
+ "language_model.model.layers.22.linear_attn.dt_bias": "model.safetensors",
449
+ "language_model.model.layers.22.linear_attn.in_proj_a.biases": "model.safetensors",
450
+ "language_model.model.layers.22.linear_attn.in_proj_a.scales": "model.safetensors",
451
+ "language_model.model.layers.22.linear_attn.in_proj_a.weight": "model.safetensors",
452
+ "language_model.model.layers.22.linear_attn.in_proj_b.biases": "model.safetensors",
453
+ "language_model.model.layers.22.linear_attn.in_proj_b.scales": "model.safetensors",
454
+ "language_model.model.layers.22.linear_attn.in_proj_b.weight": "model.safetensors",
455
+ "language_model.model.layers.22.linear_attn.in_proj_qkv.biases": "model.safetensors",
456
+ "language_model.model.layers.22.linear_attn.in_proj_qkv.scales": "model.safetensors",
457
+ "language_model.model.layers.22.linear_attn.in_proj_qkv.weight": "model.safetensors",
458
+ "language_model.model.layers.22.linear_attn.in_proj_z.biases": "model.safetensors",
459
+ "language_model.model.layers.22.linear_attn.in_proj_z.scales": "model.safetensors",
460
+ "language_model.model.layers.22.linear_attn.in_proj_z.weight": "model.safetensors",
461
+ "language_model.model.layers.22.linear_attn.norm.weight": "model.safetensors",
462
+ "language_model.model.layers.22.linear_attn.out_proj.biases": "model.safetensors",
463
+ "language_model.model.layers.22.linear_attn.out_proj.scales": "model.safetensors",
464
+ "language_model.model.layers.22.linear_attn.out_proj.weight": "model.safetensors",
465
+ "language_model.model.layers.22.mlp.down_proj.biases": "model.safetensors",
466
+ "language_model.model.layers.22.mlp.down_proj.scales": "model.safetensors",
467
+ "language_model.model.layers.22.mlp.down_proj.weight": "model.safetensors",
468
+ "language_model.model.layers.22.mlp.gate_proj.biases": "model.safetensors",
469
+ "language_model.model.layers.22.mlp.gate_proj.scales": "model.safetensors",
470
+ "language_model.model.layers.22.mlp.gate_proj.weight": "model.safetensors",
471
+ "language_model.model.layers.22.mlp.up_proj.biases": "model.safetensors",
472
+ "language_model.model.layers.22.mlp.up_proj.scales": "model.safetensors",
473
+ "language_model.model.layers.22.mlp.up_proj.weight": "model.safetensors",
474
+ "language_model.model.layers.22.post_attention_layernorm.weight": "model.safetensors",
475
+ "language_model.model.layers.23.input_layernorm.weight": "model.safetensors",
476
+ "language_model.model.layers.23.mlp.down_proj.biases": "model.safetensors",
477
+ "language_model.model.layers.23.mlp.down_proj.scales": "model.safetensors",
478
+ "language_model.model.layers.23.mlp.down_proj.weight": "model.safetensors",
479
+ "language_model.model.layers.23.mlp.gate_proj.biases": "model.safetensors",
480
+ "language_model.model.layers.23.mlp.gate_proj.scales": "model.safetensors",
481
+ "language_model.model.layers.23.mlp.gate_proj.weight": "model.safetensors",
482
+ "language_model.model.layers.23.mlp.up_proj.biases": "model.safetensors",
483
+ "language_model.model.layers.23.mlp.up_proj.scales": "model.safetensors",
484
+ "language_model.model.layers.23.mlp.up_proj.weight": "model.safetensors",
485
+ "language_model.model.layers.23.post_attention_layernorm.weight": "model.safetensors",
486
+ "language_model.model.layers.23.self_attn.k_norm.weight": "model.safetensors",
487
+ "language_model.model.layers.23.self_attn.k_proj.biases": "model.safetensors",
488
+ "language_model.model.layers.23.self_attn.k_proj.scales": "model.safetensors",
489
+ "language_model.model.layers.23.self_attn.k_proj.weight": "model.safetensors",
490
+ "language_model.model.layers.23.self_attn.o_proj.biases": "model.safetensors",
491
+ "language_model.model.layers.23.self_attn.o_proj.scales": "model.safetensors",
492
+ "language_model.model.layers.23.self_attn.o_proj.weight": "model.safetensors",
493
+ "language_model.model.layers.23.self_attn.q_norm.weight": "model.safetensors",
494
+ "language_model.model.layers.23.self_attn.q_proj.biases": "model.safetensors",
495
+ "language_model.model.layers.23.self_attn.q_proj.scales": "model.safetensors",
496
+ "language_model.model.layers.23.self_attn.q_proj.weight": "model.safetensors",
497
+ "language_model.model.layers.23.self_attn.v_proj.biases": "model.safetensors",
498
+ "language_model.model.layers.23.self_attn.v_proj.scales": "model.safetensors",
499
+ "language_model.model.layers.23.self_attn.v_proj.weight": "model.safetensors",
500
+ "language_model.model.layers.3.input_layernorm.weight": "model.safetensors",
501
+ "language_model.model.layers.3.mlp.down_proj.biases": "model.safetensors",
502
+ "language_model.model.layers.3.mlp.down_proj.scales": "model.safetensors",
503
+ "language_model.model.layers.3.mlp.down_proj.weight": "model.safetensors",
504
+ "language_model.model.layers.3.mlp.gate_proj.biases": "model.safetensors",
505
+ "language_model.model.layers.3.mlp.gate_proj.scales": "model.safetensors",
506
+ "language_model.model.layers.3.mlp.gate_proj.weight": "model.safetensors",
507
+ "language_model.model.layers.3.mlp.up_proj.biases": "model.safetensors",
508
+ "language_model.model.layers.3.mlp.up_proj.scales": "model.safetensors",
509
+ "language_model.model.layers.3.mlp.up_proj.weight": "model.safetensors",
510
+ "language_model.model.layers.3.post_attention_layernorm.weight": "model.safetensors",
511
+ "language_model.model.layers.3.self_attn.k_norm.weight": "model.safetensors",
512
+ "language_model.model.layers.3.self_attn.k_proj.biases": "model.safetensors",
513
+ "language_model.model.layers.3.self_attn.k_proj.scales": "model.safetensors",
514
+ "language_model.model.layers.3.self_attn.k_proj.weight": "model.safetensors",
515
+ "language_model.model.layers.3.self_attn.o_proj.biases": "model.safetensors",
516
+ "language_model.model.layers.3.self_attn.o_proj.scales": "model.safetensors",
517
+ "language_model.model.layers.3.self_attn.o_proj.weight": "model.safetensors",
518
+ "language_model.model.layers.3.self_attn.q_norm.weight": "model.safetensors",
519
+ "language_model.model.layers.3.self_attn.q_proj.biases": "model.safetensors",
520
+ "language_model.model.layers.3.self_attn.q_proj.scales": "model.safetensors",
521
+ "language_model.model.layers.3.self_attn.q_proj.weight": "model.safetensors",
522
+ "language_model.model.layers.3.self_attn.v_proj.biases": "model.safetensors",
523
+ "language_model.model.layers.3.self_attn.v_proj.scales": "model.safetensors",
524
+ "language_model.model.layers.3.self_attn.v_proj.weight": "model.safetensors",
525
+ "language_model.model.layers.4.input_layernorm.weight": "model.safetensors",
526
+ "language_model.model.layers.4.linear_attn.A_log": "model.safetensors",
527
+ "language_model.model.layers.4.linear_attn.conv1d.weight": "model.safetensors",
528
+ "language_model.model.layers.4.linear_attn.dt_bias": "model.safetensors",
529
+ "language_model.model.layers.4.linear_attn.in_proj_a.biases": "model.safetensors",
530
+ "language_model.model.layers.4.linear_attn.in_proj_a.scales": "model.safetensors",
531
+ "language_model.model.layers.4.linear_attn.in_proj_a.weight": "model.safetensors",
532
+ "language_model.model.layers.4.linear_attn.in_proj_b.biases": "model.safetensors",
533
+ "language_model.model.layers.4.linear_attn.in_proj_b.scales": "model.safetensors",
534
+ "language_model.model.layers.4.linear_attn.in_proj_b.weight": "model.safetensors",
535
+ "language_model.model.layers.4.linear_attn.in_proj_qkv.biases": "model.safetensors",
536
+ "language_model.model.layers.4.linear_attn.in_proj_qkv.scales": "model.safetensors",
537
+ "language_model.model.layers.4.linear_attn.in_proj_qkv.weight": "model.safetensors",
538
+ "language_model.model.layers.4.linear_attn.in_proj_z.biases": "model.safetensors",
539
+ "language_model.model.layers.4.linear_attn.in_proj_z.scales": "model.safetensors",
540
+ "language_model.model.layers.4.linear_attn.in_proj_z.weight": "model.safetensors",
541
+ "language_model.model.layers.4.linear_attn.norm.weight": "model.safetensors",
542
+ "language_model.model.layers.4.linear_attn.out_proj.biases": "model.safetensors",
543
+ "language_model.model.layers.4.linear_attn.out_proj.scales": "model.safetensors",
544
+ "language_model.model.layers.4.linear_attn.out_proj.weight": "model.safetensors",
545
+ "language_model.model.layers.4.mlp.down_proj.biases": "model.safetensors",
546
+ "language_model.model.layers.4.mlp.down_proj.scales": "model.safetensors",
547
+ "language_model.model.layers.4.mlp.down_proj.weight": "model.safetensors",
548
+ "language_model.model.layers.4.mlp.gate_proj.biases": "model.safetensors",
549
+ "language_model.model.layers.4.mlp.gate_proj.scales": "model.safetensors",
550
+ "language_model.model.layers.4.mlp.gate_proj.weight": "model.safetensors",
551
+ "language_model.model.layers.4.mlp.up_proj.biases": "model.safetensors",
552
+ "language_model.model.layers.4.mlp.up_proj.scales": "model.safetensors",
553
+ "language_model.model.layers.4.mlp.up_proj.weight": "model.safetensors",
554
+ "language_model.model.layers.4.post_attention_layernorm.weight": "model.safetensors",
555
+ "language_model.model.layers.5.input_layernorm.weight": "model.safetensors",
556
+ "language_model.model.layers.5.linear_attn.A_log": "model.safetensors",
557
+ "language_model.model.layers.5.linear_attn.conv1d.weight": "model.safetensors",
558
+ "language_model.model.layers.5.linear_attn.dt_bias": "model.safetensors",
559
+ "language_model.model.layers.5.linear_attn.in_proj_a.biases": "model.safetensors",
560
+ "language_model.model.layers.5.linear_attn.in_proj_a.scales": "model.safetensors",
561
+ "language_model.model.layers.5.linear_attn.in_proj_a.weight": "model.safetensors",
562
+ "language_model.model.layers.5.linear_attn.in_proj_b.biases": "model.safetensors",
563
+ "language_model.model.layers.5.linear_attn.in_proj_b.scales": "model.safetensors",
564
+ "language_model.model.layers.5.linear_attn.in_proj_b.weight": "model.safetensors",
565
+ "language_model.model.layers.5.linear_attn.in_proj_qkv.biases": "model.safetensors",
566
+ "language_model.model.layers.5.linear_attn.in_proj_qkv.scales": "model.safetensors",
567
+ "language_model.model.layers.5.linear_attn.in_proj_qkv.weight": "model.safetensors",
568
+ "language_model.model.layers.5.linear_attn.in_proj_z.biases": "model.safetensors",
569
+ "language_model.model.layers.5.linear_attn.in_proj_z.scales": "model.safetensors",
570
+ "language_model.model.layers.5.linear_attn.in_proj_z.weight": "model.safetensors",
571
+ "language_model.model.layers.5.linear_attn.norm.weight": "model.safetensors",
572
+ "language_model.model.layers.5.linear_attn.out_proj.biases": "model.safetensors",
573
+ "language_model.model.layers.5.linear_attn.out_proj.scales": "model.safetensors",
574
+ "language_model.model.layers.5.linear_attn.out_proj.weight": "model.safetensors",
575
+ "language_model.model.layers.5.mlp.down_proj.biases": "model.safetensors",
576
+ "language_model.model.layers.5.mlp.down_proj.scales": "model.safetensors",
577
+ "language_model.model.layers.5.mlp.down_proj.weight": "model.safetensors",
578
+ "language_model.model.layers.5.mlp.gate_proj.biases": "model.safetensors",
579
+ "language_model.model.layers.5.mlp.gate_proj.scales": "model.safetensors",
580
+ "language_model.model.layers.5.mlp.gate_proj.weight": "model.safetensors",
581
+ "language_model.model.layers.5.mlp.up_proj.biases": "model.safetensors",
582
+ "language_model.model.layers.5.mlp.up_proj.scales": "model.safetensors",
583
+ "language_model.model.layers.5.mlp.up_proj.weight": "model.safetensors",
584
+ "language_model.model.layers.5.post_attention_layernorm.weight": "model.safetensors",
585
+ "language_model.model.layers.6.input_layernorm.weight": "model.safetensors",
586
+ "language_model.model.layers.6.linear_attn.A_log": "model.safetensors",
587
+ "language_model.model.layers.6.linear_attn.conv1d.weight": "model.safetensors",
588
+ "language_model.model.layers.6.linear_attn.dt_bias": "model.safetensors",
589
+ "language_model.model.layers.6.linear_attn.in_proj_a.biases": "model.safetensors",
590
+ "language_model.model.layers.6.linear_attn.in_proj_a.scales": "model.safetensors",
591
+ "language_model.model.layers.6.linear_attn.in_proj_a.weight": "model.safetensors",
592
+ "language_model.model.layers.6.linear_attn.in_proj_b.biases": "model.safetensors",
593
+ "language_model.model.layers.6.linear_attn.in_proj_b.scales": "model.safetensors",
594
+ "language_model.model.layers.6.linear_attn.in_proj_b.weight": "model.safetensors",
595
+ "language_model.model.layers.6.linear_attn.in_proj_qkv.biases": "model.safetensors",
596
+ "language_model.model.layers.6.linear_attn.in_proj_qkv.scales": "model.safetensors",
597
+ "language_model.model.layers.6.linear_attn.in_proj_qkv.weight": "model.safetensors",
598
+ "language_model.model.layers.6.linear_attn.in_proj_z.biases": "model.safetensors",
599
+ "language_model.model.layers.6.linear_attn.in_proj_z.scales": "model.safetensors",
600
+ "language_model.model.layers.6.linear_attn.in_proj_z.weight": "model.safetensors",
601
+ "language_model.model.layers.6.linear_attn.norm.weight": "model.safetensors",
602
+ "language_model.model.layers.6.linear_attn.out_proj.biases": "model.safetensors",
603
+ "language_model.model.layers.6.linear_attn.out_proj.scales": "model.safetensors",
604
+ "language_model.model.layers.6.linear_attn.out_proj.weight": "model.safetensors",
605
+ "language_model.model.layers.6.mlp.down_proj.biases": "model.safetensors",
606
+ "language_model.model.layers.6.mlp.down_proj.scales": "model.safetensors",
607
+ "language_model.model.layers.6.mlp.down_proj.weight": "model.safetensors",
608
+ "language_model.model.layers.6.mlp.gate_proj.biases": "model.safetensors",
609
+ "language_model.model.layers.6.mlp.gate_proj.scales": "model.safetensors",
610
+ "language_model.model.layers.6.mlp.gate_proj.weight": "model.safetensors",
611
+ "language_model.model.layers.6.mlp.up_proj.biases": "model.safetensors",
612
+ "language_model.model.layers.6.mlp.up_proj.scales": "model.safetensors",
613
+ "language_model.model.layers.6.mlp.up_proj.weight": "model.safetensors",
614
+ "language_model.model.layers.6.post_attention_layernorm.weight": "model.safetensors",
615
+ "language_model.model.layers.7.input_layernorm.weight": "model.safetensors",
616
+ "language_model.model.layers.7.mlp.down_proj.biases": "model.safetensors",
617
+ "language_model.model.layers.7.mlp.down_proj.scales": "model.safetensors",
618
+ "language_model.model.layers.7.mlp.down_proj.weight": "model.safetensors",
619
+ "language_model.model.layers.7.mlp.gate_proj.biases": "model.safetensors",
620
+ "language_model.model.layers.7.mlp.gate_proj.scales": "model.safetensors",
621
+ "language_model.model.layers.7.mlp.gate_proj.weight": "model.safetensors",
622
+ "language_model.model.layers.7.mlp.up_proj.biases": "model.safetensors",
623
+ "language_model.model.layers.7.mlp.up_proj.scales": "model.safetensors",
624
+ "language_model.model.layers.7.mlp.up_proj.weight": "model.safetensors",
625
+ "language_model.model.layers.7.post_attention_layernorm.weight": "model.safetensors",
626
+ "language_model.model.layers.7.self_attn.k_norm.weight": "model.safetensors",
627
+ "language_model.model.layers.7.self_attn.k_proj.biases": "model.safetensors",
628
+ "language_model.model.layers.7.self_attn.k_proj.scales": "model.safetensors",
629
+ "language_model.model.layers.7.self_attn.k_proj.weight": "model.safetensors",
630
+ "language_model.model.layers.7.self_attn.o_proj.biases": "model.safetensors",
631
+ "language_model.model.layers.7.self_attn.o_proj.scales": "model.safetensors",
632
+ "language_model.model.layers.7.self_attn.o_proj.weight": "model.safetensors",
633
+ "language_model.model.layers.7.self_attn.q_norm.weight": "model.safetensors",
634
+ "language_model.model.layers.7.self_attn.q_proj.biases": "model.safetensors",
635
+ "language_model.model.layers.7.self_attn.q_proj.scales": "model.safetensors",
636
+ "language_model.model.layers.7.self_attn.q_proj.weight": "model.safetensors",
637
+ "language_model.model.layers.7.self_attn.v_proj.biases": "model.safetensors",
638
+ "language_model.model.layers.7.self_attn.v_proj.scales": "model.safetensors",
639
+ "language_model.model.layers.7.self_attn.v_proj.weight": "model.safetensors",
640
+ "language_model.model.layers.8.input_layernorm.weight": "model.safetensors",
641
+ "language_model.model.layers.8.linear_attn.A_log": "model.safetensors",
642
+ "language_model.model.layers.8.linear_attn.conv1d.weight": "model.safetensors",
643
+ "language_model.model.layers.8.linear_attn.dt_bias": "model.safetensors",
644
+ "language_model.model.layers.8.linear_attn.in_proj_a.biases": "model.safetensors",
645
+ "language_model.model.layers.8.linear_attn.in_proj_a.scales": "model.safetensors",
646
+ "language_model.model.layers.8.linear_attn.in_proj_a.weight": "model.safetensors",
647
+ "language_model.model.layers.8.linear_attn.in_proj_b.biases": "model.safetensors",
648
+ "language_model.model.layers.8.linear_attn.in_proj_b.scales": "model.safetensors",
649
+ "language_model.model.layers.8.linear_attn.in_proj_b.weight": "model.safetensors",
650
+ "language_model.model.layers.8.linear_attn.in_proj_qkv.biases": "model.safetensors",
651
+ "language_model.model.layers.8.linear_attn.in_proj_qkv.scales": "model.safetensors",
652
+ "language_model.model.layers.8.linear_attn.in_proj_qkv.weight": "model.safetensors",
653
+ "language_model.model.layers.8.linear_attn.in_proj_z.biases": "model.safetensors",
654
+ "language_model.model.layers.8.linear_attn.in_proj_z.scales": "model.safetensors",
655
+ "language_model.model.layers.8.linear_attn.in_proj_z.weight": "model.safetensors",
656
+ "language_model.model.layers.8.linear_attn.norm.weight": "model.safetensors",
657
+ "language_model.model.layers.8.linear_attn.out_proj.biases": "model.safetensors",
658
+ "language_model.model.layers.8.linear_attn.out_proj.scales": "model.safetensors",
659
+ "language_model.model.layers.8.linear_attn.out_proj.weight": "model.safetensors",
660
+ "language_model.model.layers.8.mlp.down_proj.biases": "model.safetensors",
661
+ "language_model.model.layers.8.mlp.down_proj.scales": "model.safetensors",
662
+ "language_model.model.layers.8.mlp.down_proj.weight": "model.safetensors",
663
+ "language_model.model.layers.8.mlp.gate_proj.biases": "model.safetensors",
664
+ "language_model.model.layers.8.mlp.gate_proj.scales": "model.safetensors",
665
+ "language_model.model.layers.8.mlp.gate_proj.weight": "model.safetensors",
666
+ "language_model.model.layers.8.mlp.up_proj.biases": "model.safetensors",
667
+ "language_model.model.layers.8.mlp.up_proj.scales": "model.safetensors",
668
+ "language_model.model.layers.8.mlp.up_proj.weight": "model.safetensors",
669
+ "language_model.model.layers.8.post_attention_layernorm.weight": "model.safetensors",
670
+ "language_model.model.layers.9.input_layernorm.weight": "model.safetensors",
671
+ "language_model.model.layers.9.linear_attn.A_log": "model.safetensors",
672
+ "language_model.model.layers.9.linear_attn.conv1d.weight": "model.safetensors",
673
+ "language_model.model.layers.9.linear_attn.dt_bias": "model.safetensors",
674
+ "language_model.model.layers.9.linear_attn.in_proj_a.biases": "model.safetensors",
675
+ "language_model.model.layers.9.linear_attn.in_proj_a.scales": "model.safetensors",
676
+ "language_model.model.layers.9.linear_attn.in_proj_a.weight": "model.safetensors",
677
+ "language_model.model.layers.9.linear_attn.in_proj_b.biases": "model.safetensors",
678
+ "language_model.model.layers.9.linear_attn.in_proj_b.scales": "model.safetensors",
679
+ "language_model.model.layers.9.linear_attn.in_proj_b.weight": "model.safetensors",
680
+ "language_model.model.layers.9.linear_attn.in_proj_qkv.biases": "model.safetensors",
681
+ "language_model.model.layers.9.linear_attn.in_proj_qkv.scales": "model.safetensors",
682
+ "language_model.model.layers.9.linear_attn.in_proj_qkv.weight": "model.safetensors",
683
+ "language_model.model.layers.9.linear_attn.in_proj_z.biases": "model.safetensors",
684
+ "language_model.model.layers.9.linear_attn.in_proj_z.scales": "model.safetensors",
685
+ "language_model.model.layers.9.linear_attn.in_proj_z.weight": "model.safetensors",
686
+ "language_model.model.layers.9.linear_attn.norm.weight": "model.safetensors",
687
+ "language_model.model.layers.9.linear_attn.out_proj.biases": "model.safetensors",
688
+ "language_model.model.layers.9.linear_attn.out_proj.scales": "model.safetensors",
689
+ "language_model.model.layers.9.linear_attn.out_proj.weight": "model.safetensors",
690
+ "language_model.model.layers.9.mlp.down_proj.biases": "model.safetensors",
691
+ "language_model.model.layers.9.mlp.down_proj.scales": "model.safetensors",
692
+ "language_model.model.layers.9.mlp.down_proj.weight": "model.safetensors",
693
+ "language_model.model.layers.9.mlp.gate_proj.biases": "model.safetensors",
694
+ "language_model.model.layers.9.mlp.gate_proj.scales": "model.safetensors",
695
+ "language_model.model.layers.9.mlp.gate_proj.weight": "model.safetensors",
696
+ "language_model.model.layers.9.mlp.up_proj.biases": "model.safetensors",
697
+ "language_model.model.layers.9.mlp.up_proj.scales": "model.safetensors",
698
+ "language_model.model.layers.9.mlp.up_proj.weight": "model.safetensors",
699
+ "language_model.model.layers.9.post_attention_layernorm.weight": "model.safetensors",
700
+ "language_model.model.norm.weight": "model.safetensors"
701
+ }
702
+ }
8bit/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
8bit/tokenizer_config.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "tool_parser_type": "qwen3_coder",
29
+ "unk_token": null,
30
+ "video_token": "<|video_pad|>",
31
+ "vision_bos_token": "<|vision_start|>",
32
+ "vision_eos_token": "<|vision_end|>"
33
+ }
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX
2
+ Copyright 2026 chaoliangUNSW. Licensed under the Apache License, Version 2.0 (see LICENSE).
3
+
4
+ This model is a fine-tuned derivative of Qwen3.5-2B (https://huggingface.co/Qwen/Qwen3.5-2B,
5
+ revision 15852e8c16360a2fea060d615a32b45270f8a8fc), Copyright 2026 Alibaba Cloud, licensed under the Apache
6
+ License, Version 2.0. The LICENSE file in this repository is the license file distributed with Qwen3.5-2B.
7
+
8
+ Modifications relative to Qwen3.5-2B:
9
+ - all text-model weights were fine-tuned (full fine-tuning) to score typed decision questions
10
+ (choice / score / true-false) with a verdict readout: logit(" yes") - logit(" no") at one " ->" slot per
11
+ option, under a block-causal attention protocol (2,048-token blocks); one global calibration temperature was
12
+ fitted afterwards (readout_config.json);
13
+ - the vision tower (model.visual.*) and the multi-token-prediction head (mtp.*) were removed; the checkpoint
14
+ is a text-only Qwen3_5ForCausalLM with tied input/output embeddings;
15
+ - converted to MLX format with mlx-lm 0.31.3 / mlx 0.32.2 (https://github.com/ml-explore/mlx-lm,
16
+ MIT License, Copyright Apple Inc.): bf16/ = bfloat16, 8bit/ = affine 8-bit quantisation, group size 64;
17
+ each folder also holds macjev_norms_fp32.safetensors (the exact FP32 (1 + w) RMSNorm weights);
18
+ - the runtime script contains a modified copy of mlx-lm 0.31.3's qwen3_5 GatedDeltaNet.__call__ (MIT License,
19
+ Copyright Apple Inc.; the q/k normalisation epsilon changed and the sharding branches removed); the MIT
20
+ copyright and permission notice is reproduced in THIRD_PARTY_NOTICES.md;
21
+ - added the runtime script, readout/release configuration files, the integrity manifest and this NOTICE.
22
+
23
+ The question types (choice / score / noul) follow the typed-decision convention of Laya
24
+ (https://github.com/NandhaKishorM/laya, Apache-2.0) so both models can be evaluated on the same
25
+ inputs. No Laya code or weights are included.
26
+
27
+ Part of the third generation (v3) of the Jev-Style decision series (v3 also includes
28
+ chaoliangUNSW/Jev-Style-0.8B-Decision-v3). Earlier generations: v1 = chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision
29
+ (public GGUF release: chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF) and v2 =
30
+ chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2, both built on Qwen3.5-2B-Base. Jev-Style-2B-Decision-v3 is
31
+ a full fine-tune of Qwen/Qwen3.5-2B.
32
+
33
+ Not affiliated with, endorsed by or connected to TypeSafe or Jev. "Jev-Style" only describes the kind
34
+ of model (a small typed-decision model in a similar style); no Jev weights, code or outputs are included.
35
+ Not affiliated with or endorsed by Alibaba Cloud / the Qwen team or the Laya authors.
README.md ADDED
@@ -0,0 +1,194 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model: chaoliangUNSW/Jev-Style-2B-Decision-v3
4
+ base_model_relation: quantized
5
+ library_name: mlx
6
+ pipeline_tag: text-classification
7
+ tags:
8
+ - decision-model
9
+ - decision-making
10
+ - jev-style
11
+ - system-one
12
+ - calibration
13
+ - long-context
14
+ - qwen3.5
15
+ - mlx
16
+ - apple-silicon
17
+ - on-device
18
+ - llm-routing
19
+ - guardrails
20
+ ---
21
+
22
+ # Jev-Style-2B-Decision-v3-MLX
23
+
24
+ **[Try it in your browser →](https://huggingface.co/spaces/chaoliangUNSW/jev-style-2b)**
25
+
26
+ **Website:** [jevstyle.com](https://jevstyle.com/#v3-2b) · **GitHub:** [jev-style](https://github.com/lawrence3699/jev-style) · **Collection:** [all v3 builds and demos](https://huggingface.co/collections/chaoliangUNSW/jev-style-decision-v3-08b-2b-6ab87f32380cbd8c03b608b9)
27
+
28
+ **Jev-Style decision series:** [v1 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-MLX-bf16) → [v2 · 2B](https://huggingface.co/chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2-MLX-bf16) → [v3 · 0.8B](https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3-MLX) → **v3 · 2B**
29
+
30
+ <!-- PIP_SNIPPET_AFTER_0.3.0 -->
31
+
32
+ **Jev-style decisions, now at 2B, native on Apple silicon.** These are the MLX builds of
33
+ [Jev-Style-2B-Decision-v3](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3): bf16 (3.76 GB) and
34
+ 8-bit (2.00 GB) in one repository, with one runtime (`jev_style_decision_mlx.py`; choose the folder with `--model-dir bf16` or `--model-dir 8bit`). Full
35
+ results, protocols, training data and licences are on the
36
+ [main model card](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3).
37
+
38
+ ![Jev-Style 2B Decision v3: 73.6% on JevBench v1.4.1 public items, the highest among the Qwen3.5-2B-family systems on the board; Jev is ahead at 86.6%; shown are the Qwen3.5-2B-family systems, our 0.8B v3, Laya and Jev, and 42 of the 82 board systems score higher; 25,600 tokens per call with no option cap](figures/banner.png)
39
+
40
+ | Public benchmark | **Jev-Style v3 · 2B** | Jev-Style v3 · 0.8B | Jev 1.13 (API) |
41
+ |---|:---:|:---:|:---:|
42
+ | JevBench v1.4.1, 231 public items ↑ | 73.6% | 64.1% | 86.6% |
43
+ | tweet_topic, zero-shot, accuracy ↑ | 82.2% | 75.5% | 79.3%¹ |
44
+ | fin_topic, zero-shot, accuracy ↑ | 61.1% | 46.7% | 67.0%¹ |
45
+ | Longest input per call | 25,600 tokens, no option cap | 25,600 tokens | |
46
+
47
+ <sub>Benchmark numbers of the model, measured with its GGUF F16 build (the pre-declared engine; one global temperature; each benchmark run once). The MLX builds were checked against the same FP32 reference (table below). JevBench: self-run with the official harness, not an official board entry; 95% CI 67.6–78.9% (Wilson). 73.6% is the highest JevBench public accuracy among the Qwen3.5-2B-family systems on the v1.4.1 board (decider-2b 71.0%, open-jev-zefan-2b 64.5%); decider-2b lies inside the CI. Jev is well ahead on JevBench and ahead on fin_topic. ¹ Jev numbers from the elcronos study (raw API), not re-run by us; on tweet_topic the 2B's macro-F1 (67.8%) is below Jev's (69.4%). Details: [main card](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3#results).</sub>
48
+
49
+ **25,600 tokens, no option cap.** State, questions and every option share one 25,600-token budget; question and
50
+ options over 2,048 tokens use a numbered-option catalogue. Nothing is truncated.
51
+
52
+ ## Precisions and parity
53
+
54
+ | Precision | Folder | Weights | Same top-1 as PyTorch FP32 (1,000 rows) | Max abs Δp | Accuracy (FP32: 80.8%) | Long fixture (43 questions) | Gate |
55
+ |---|---|---:|---:|---:|---:|---:|:---:|
56
+ | **bf16** (default) | `bf16/` | 3.76 GB | **99.7%** | 0.035 | 80.5% | **43 / 43** | PASS |
57
+ | **8-bit** (affine, group size 64) | `8bit/` | 2.00 GB | **99.6%** | 0.162 | 80.8% | **43 / 43** | PASS |
58
+
59
+ - Each folder holds `model.safetensors`, its `config.json`, the tokenizer and `macjev_norms_fp32.safetensors`
60
+ (FP32 norm weights, 0.42 MB, required). The top level holds the shared runtime `jev_style_decision_mlx.py`,
61
+ `readout_config.json`, `release_config.json`, the sha256 `manifest.json`, `LICENSE`, `NOTICE`,
62
+ `THIRD_PARTY_NOTICES.md` and a copy of `bf16/config.json` (for the Hub's download counter). Converted with mlx
63
+ 0.32.2 and mlx-lm 0.31.3.
64
+ - The 8-bit build makes the same call as full precision on 99.6% of the 1,000 rows at about half the size.
65
+ - **Which folder:** `bf16/` for closeness to the FP32 reference (max abs Δp 0.035 vs 0.162); `8bit/` when memory is
66
+ tight.
67
+
68
+ <sub>Reference: HF FP32 on CPU, exact block attention, on the released bf16 checkpoint; gates declared before any format was scored. 1,000 real development rows (≤4,096 tokens) test agreement between formats. Long fixture: 35 requests / 43 questions up to 25,600 tokens, including catalogue-overflow questions and up to 151 options. Sizes are the weight files (GB = 10^9 bytes).</sub>
69
+
70
+ <!-- LATENCY:BEGIN generated from latency_2b.json (validation/latency_2b.json in the main repository); do not edit by hand -->
71
+
72
+ ## Speed
73
+
74
+ **Read once, then ask.** On an Apple M1 Max (MLX bf16), the first question about a 24,501-token input took 15.5 s; a further question about the same state took 0.15 s, because the state is computed once and reused (medians). Ten questions about that state in one call took 16.2 s.
75
+
76
+ | State | Questions per call | MLX bf16 | MLX 8-bit |
77
+ |---|---:|---:|---:|
78
+ | 878 tokens | 1 | 0.57 s | 0.72 s |
79
+ | 878 tokens | 10 | 1.07 s | 1.27 s |
80
+ | 3,950 tokens | 1 | 2.26 s | 2.96 s |
81
+ | 3,950 tokens | 10 | 2.80 s | 3.54 s |
82
+ | 24,436 tokens | 1 | 15.5 s | 19.9 s |
83
+ | 24,436 tokens | 10 | 16.2 s | 20.8 s |
84
+ | 24,436 tokens, already computed | 1 | 0.15 s | 0.16 s |
85
+
86
+ - Pick a precision folder by size and closeness to FP32 (Precisions table above). bf16 and 8-bit were timed one after another on the same shared machine; in these runs 8-bit was the slower folder in every row (indicative only).
87
+
88
+ <sub>Apple M1 Max, 64 GB, macOS 15.7.5. Wall time around one `decide` / `score_many` call (tokenisation included), median of 3 calls with the state recomputed each time; the model was loaded beforehand (loading took 3.1–3.7 s here, not included). States: English documentation and source code of 878, 3,950, 24,436 tokens plus the question; 10 questions = 4 choice, 4 true/false and 2 score questions about the same state in one call; with the question and options each input was up to 943, 4,015 and 24,501 tokens. MLX: mlx 0.32.2 / mlx-lm 0.31.3. Results were identical with and without a precomputed state. Other jobs shared the machine during these runs (1-minute load average 5.5–8.6 at the end of each run), so treat the numbers as indicative. All rows here were measured in one session (2026-09-27 03:21–03:28 AEST); an earlier run of the same rows (00:37–00:45 AEST, load average 11.8–27.5) was discarded because other jobs had slowed it (its times were up to 2.1× longer).</sub>
89
+
90
+ <!-- LATENCY:END -->
91
+
92
+ ## Quick start
93
+
94
+ ```bash
95
+ pip install -U huggingface_hub
96
+ hf download chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX --local-dir Jev-Style-2B-Decision-v3-MLX
97
+ cd Jev-Style-2B-Decision-v3-MLX
98
+ pip install -r requirements.txt # mlx 0.32.2, mlx-lm 0.31.3 (exactly), transformers 5.17.0, tokenizers, numpy
99
+ ```
100
+
101
+ Only one precision is needed: add `--exclude "8bit/*"` to the download for bf16 only, or `--exclude "bf16/*"` for
102
+ 8-bit only. Pass the precision folder to the runtime with `--model-dir bf16` or `--model-dir 8bit` (in Python,
103
+ `JevStyleDecisionMLX("bf16")` or `JevStyleDecisionMLX("8bit")`).
104
+
105
+ ```bash
106
+ # choice question, bf16 weights
107
+ python jev_style_decision_mlx.py --model-dir bf16 --state "Finder is open on ~/Reports. The user asked: email report.pdf to Ana." \
108
+ --question "What should the agent do next?" \
109
+ --options '{"attach_report": "attach report.pdf to a new email to Ana", "rename_report": null, "close_finder": "close the Finder window"}'
110
+
111
+ # true/false question about a JSON state, 8-bit weights
112
+ python jev_style_decision_mlx.py --model-dir 8bit \
113
+ --state-json '{"app": "Mail", "draft": {"to": "ana@example.com", "attachments": []}}' \
114
+ --question "Is the report attached?" --qtype noul
115
+
116
+ # score (ordinal) question, uncalibrated probabilities (T = 1.0)
117
+ python jev_style_decision_mlx.py --model-dir 8bit --state "The deploy failed twice with the same migration error." \
118
+ --question "How risky is retrying now?" --qtype score --options '["no risk", "some risk", "high risk"]' --temperature 1.0
119
+
120
+ # full typed question as JSON, first checking the sha256 of the runtime files and the chosen precision folder (manifest.json)
121
+ python jev_style_decision_mlx.py --model-dir bf16 --verify --state "cart: 3 items" \
122
+ --question '{"t": "choice", "ins": "Next step?", "crit": {"checkout": null, "keep_shopping": null}}'
123
+ ```
124
+
125
+ The output has `answer`, `probabilities` and raw `scores` per option, `temperature` (0.8279, the calibrated global
126
+ temperature), `top_probability`, `entropy_concentration`, `input_tokens`, `state_tokens`, `head_tokens`, `blocks`
127
+ and `catalogue_overflow`.
128
+
129
+ ```python
130
+ from jev_style_decision_mlx import JevStyleDecisionMLX
131
+ m = JevStyleDecisionMLX("8bit") # the 8-bit precision folder
132
+ state = {"screen": "Settings > Wi-Fi", "goal": "join the network Office-5G"}
133
+ qs = [{"t": "choice", "ins": "Next action?", "crit": {"tap_office_5g": None, "toggle_wifi_off": None, "go_back": None}},
134
+ {"t": "noul", "ins": "Is Wi-Fi turned on?", "crit": None}]
135
+ for r in m.score_many(state, qs): # the state is computed once and reused for every question
136
+ print(r["answer"], round(r["top_probability"], 4))
137
+ m.decide(state, qs[0]) # same state again: reused (m.last_timing["state_reused"] is True)
138
+ m.close()
139
+ ```
140
+
141
+ - Batch mode: `--jsonl requests.jsonl` (or `-` for stdin), rows `{"id"?, "state", "question", "options"?, "qtype"?,
142
+ "temperature"?}`; consecutive rows with an identical state share one state computation, and a malformed row gets
143
+ an `"error"` field while the batch continues.
144
+ - **Long inputs.** The complete input can be up to 25,600 tokens, with no separate question/options cap. In our
145
+ smoke test a 22,005-token state with 100 described options ran as 25,393 tokens in 14 blocks, using the
146
+ catalogue layout. A larger input raises `InputBudgetError` (CLI exit status 2); nothing is truncated.
147
+ - **mlx-lm must be exactly 0.31.3.** The runtime corrects two Qwen3.5 numerics inside mlx-lm and checks the source
148
+ it patches; other versions are refused with the install command. It also refuses to run without
149
+ `macjev_norms_fp32.safetensors` or `readout_config.json`, and never falls back to stock norms or T = 1.
150
+ - The input format, block attention and the readout are described on the
151
+ [main card](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3#input-format-and-readout).
152
+
153
+ ## Scope and limits
154
+
155
+ - **Runtime required.** `mlx_lm.generate` can load the weights in `bf16/` or `8bit/` (the
156
+ repository root holds no weights) but still cannot produce the decision scores, and it would run causal attention.
157
+ Use `jev_style_decision_mlx.py`.
158
+ - **25,600 tokens** is the limit for the whole input (state + question + options + readout).
159
+ - **Reduced-data training.** The model was trained on a reduced data pool (60M tokens).
160
+ - **Decision Index.** Not run by us. The training pool includes the train splits of 7 of its benchmarks and
161
+ format-imitating data for 7 more, so results on these are not zero-shot; the 14 benchmarks that are not zero-shot
162
+ for this model are named on the [main card](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3#scope-and-limits).
163
+
164
+ <details>
165
+ <summary><strong>Results charts</strong></summary>
166
+
167
+ ![JevBench v1.4.1 public accuracy: 2B v3 vs the Qwen3.5-2B-family systems, 0.8B v3 and Laya, with Jev as a reference line](figures/jevbench.png)
168
+
169
+ ![Zero-shot tweet_topic and fin_topic accuracy: 2B v3 vs 0.8B v3 and Jev](figures/zeroshot.png)
170
+
171
+ <sub>Protocol notes are under each chart and on the main card. Plotted values and sources: [jevbench.data.json](figures/jevbench.data.json), [zeroshot.data.json](figures/zeroshot.data.json).</sub>
172
+
173
+ </details>
174
+
175
+ ## Disclaimers and licence
176
+
177
+ Apache-2.0. Built on Qwen/Qwen3.5-2B (Apache-2.0); `NOTICE` lists the modifications. Some training data has
178
+ restrictive or unclear terms (for example research-only jailbreak prompts and share-alike CC BY-SA sources), and some
179
+ training rows are outputs of OpenAI GPT and Anthropic Claude models, whose providers' terms of use may restrict how
180
+ models trained on them may be used. See [Training data and
181
+ licences](https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3#training-data-and-licences) on the main card.
182
+ The runtime contains a modified copy of one mlx-lm 0.31.3 function (the Qwen3.5 `GatedDeltaNet.__call__`; MIT License,
183
+ Copyright Apple Inc.); its copyright and permission notice is in `THIRD_PARTY_NOTICES.md`. Not affiliated with,
184
+ endorsed by or connected to TypeSafe AI or Jev (no Jev weights, code or outputs are used), the Laya authors or the
185
+ Qwen team.
186
+
187
+ **AI disclosure:** code written with AI coding assistants (Claude Code) under my direction; I designed the project,
188
+ trained the models and verified the results.
189
+
190
+ ## Contact
191
+
192
+ I welcome internship, employment, and research collaboration opportunities. Please contact me at [**yanchaoliang369@gmail.com**](mailto:yanchaoliang369@gmail.com).
193
+
194
+ 欢迎提供实习、工作及科研合作机会,请邮件联系:[yanchaoliang369@gmail.com](mailto:yanchaoliang369@gmail.com)。
THIRD_PARTY_NOTICES.md ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party notices
2
+
3
+ This file covers third-party code that is **contained in** this repository. Libraries that are only imported at
4
+ run time (mlx, mlx-lm, tokenizers, numpy) are not redistributed here and keep their own licences.
5
+
6
+ ## mlx-lm 0.31.3 (MIT License)
7
+
8
+ `jev_style_decision_mlx.py` contains a modified copy of `GatedDeltaNet.__call__` from
9
+ `mlx_lm/models/qwen3_5.py` of mlx-lm 0.31.3 (https://github.com/ml-explore/mlx-lm; that source file carries the
10
+ header "Copyright © 2026 Apple Inc."). In the runtime it is the function `_gdn_hf_call`. Changes made:
11
+
12
+ - the q/k normalisation epsilon: `mx.fast.rms_norm(x, None, 1e-6)` became `mx.fast.rms_norm(x, None, 1e-6 / Dk)`,
13
+ so that q/k are l2-normalised with eps 1e-6 as in Hugging Face transformers / llama.cpp;
14
+ - the tensor-parallel (sharding) branches were removed: a sharded layer is refused with an error;
15
+ - rewritten as a module-level function (installed per model instance, never globally) and reformatted.
16
+
17
+ The mlx-lm licence (the LICENSE file distributed with mlx-lm 0.31.3), reproduced verbatim:
18
+
19
+ ```text
20
+ MIT License
21
+
22
+ Copyright © 2023 Apple Inc.
23
+
24
+ Permission is hereby granted, free of charge, to any person obtaining a copy
25
+ of this software and associated documentation files (the "Software"), to deal
26
+ in the Software without restriction, including without limitation the rights
27
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
28
+ copies of the Software, and to permit persons to whom the Software is
29
+ furnished to do so, subject to the following conditions:
30
+
31
+ The above copyright notice and this permission notice shall be included in all
32
+ copies or substantial portions of the Software.
33
+
34
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
35
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
36
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
37
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
38
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
39
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
40
+ SOFTWARE.
41
+ ```
bf16/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is true %}
150
+ {{- '<think>\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n\n</think>\n\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
bf16/config.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 2048,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 6144,
17
+ "layer_types": [
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "linear_attention",
21
+ "full_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention"
42
+ ],
43
+ "linear_conv_kernel_dim": 4,
44
+ "linear_key_head_dim": 128,
45
+ "linear_num_key_heads": 16,
46
+ "linear_num_value_heads": 16,
47
+ "linear_value_head_dim": 128,
48
+ "mamba_ssm_dtype": "float32",
49
+ "max_position_embeddings": 262144,
50
+ "mlp_only_layers": [],
51
+ "model_type": "qwen3_5",
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_hidden_layers": 24,
56
+ "num_key_value_heads": 2,
57
+ "pad_token_id": null,
58
+ "partial_rotary_factor": 0.25,
59
+ "rms_norm_eps": 1e-06,
60
+ "rope_parameters": {
61
+ "mrope_interleaved": true,
62
+ "mrope_section": [
63
+ 11,
64
+ 11,
65
+ 10
66
+ ],
67
+ "partial_rotary_factor": 0.25,
68
+ "rope_theta": 10000000,
69
+ "type": "default"
70
+ },
71
+ "tie_word_embeddings": true,
72
+ "transformers_version": "5.17.0",
73
+ "use_cache": false,
74
+ "vocab_size": 248320
75
+ }
bf16/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "eos_token_id": 248044,
4
+ "transformers_version": "5.17.0",
5
+ "use_cache": true
6
+ }
bf16/macjev_norms_fp32.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a30e1a72d02598b405978c0eff08c3bfca3b523fa0b3d4b4c80d2a185e0b8081
3
+ size 420112
bf16/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:00629c8d17ea115c21c3fc0d6ddc5d4aa02887343a8ff3b3cf79ab197624d574
3
+ size 3763691755
bf16/model.safetensors.index.json ADDED
@@ -0,0 +1,328 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 3763650176,
4
+ "total_parameters": 1881824512
5
+ },
6
+ "weight_map": {
7
+ "language_model.model.embed_tokens.weight": "model.safetensors",
8
+ "language_model.model.layers.0.input_layernorm.weight": "model.safetensors",
9
+ "language_model.model.layers.0.linear_attn.A_log": "model.safetensors",
10
+ "language_model.model.layers.0.linear_attn.conv1d.weight": "model.safetensors",
11
+ "language_model.model.layers.0.linear_attn.dt_bias": "model.safetensors",
12
+ "language_model.model.layers.0.linear_attn.in_proj_a.weight": "model.safetensors",
13
+ "language_model.model.layers.0.linear_attn.in_proj_b.weight": "model.safetensors",
14
+ "language_model.model.layers.0.linear_attn.in_proj_qkv.weight": "model.safetensors",
15
+ "language_model.model.layers.0.linear_attn.in_proj_z.weight": "model.safetensors",
16
+ "language_model.model.layers.0.linear_attn.norm.weight": "model.safetensors",
17
+ "language_model.model.layers.0.linear_attn.out_proj.weight": "model.safetensors",
18
+ "language_model.model.layers.0.mlp.down_proj.weight": "model.safetensors",
19
+ "language_model.model.layers.0.mlp.gate_proj.weight": "model.safetensors",
20
+ "language_model.model.layers.0.mlp.up_proj.weight": "model.safetensors",
21
+ "language_model.model.layers.0.post_attention_layernorm.weight": "model.safetensors",
22
+ "language_model.model.layers.1.input_layernorm.weight": "model.safetensors",
23
+ "language_model.model.layers.1.linear_attn.A_log": "model.safetensors",
24
+ "language_model.model.layers.1.linear_attn.conv1d.weight": "model.safetensors",
25
+ "language_model.model.layers.1.linear_attn.dt_bias": "model.safetensors",
26
+ "language_model.model.layers.1.linear_attn.in_proj_a.weight": "model.safetensors",
27
+ "language_model.model.layers.1.linear_attn.in_proj_b.weight": "model.safetensors",
28
+ "language_model.model.layers.1.linear_attn.in_proj_qkv.weight": "model.safetensors",
29
+ "language_model.model.layers.1.linear_attn.in_proj_z.weight": "model.safetensors",
30
+ "language_model.model.layers.1.linear_attn.norm.weight": "model.safetensors",
31
+ "language_model.model.layers.1.linear_attn.out_proj.weight": "model.safetensors",
32
+ "language_model.model.layers.1.mlp.down_proj.weight": "model.safetensors",
33
+ "language_model.model.layers.1.mlp.gate_proj.weight": "model.safetensors",
34
+ "language_model.model.layers.1.mlp.up_proj.weight": "model.safetensors",
35
+ "language_model.model.layers.1.post_attention_layernorm.weight": "model.safetensors",
36
+ "language_model.model.layers.10.input_layernorm.weight": "model.safetensors",
37
+ "language_model.model.layers.10.linear_attn.A_log": "model.safetensors",
38
+ "language_model.model.layers.10.linear_attn.conv1d.weight": "model.safetensors",
39
+ "language_model.model.layers.10.linear_attn.dt_bias": "model.safetensors",
40
+ "language_model.model.layers.10.linear_attn.in_proj_a.weight": "model.safetensors",
41
+ "language_model.model.layers.10.linear_attn.in_proj_b.weight": "model.safetensors",
42
+ "language_model.model.layers.10.linear_attn.in_proj_qkv.weight": "model.safetensors",
43
+ "language_model.model.layers.10.linear_attn.in_proj_z.weight": "model.safetensors",
44
+ "language_model.model.layers.10.linear_attn.norm.weight": "model.safetensors",
45
+ "language_model.model.layers.10.linear_attn.out_proj.weight": "model.safetensors",
46
+ "language_model.model.layers.10.mlp.down_proj.weight": "model.safetensors",
47
+ "language_model.model.layers.10.mlp.gate_proj.weight": "model.safetensors",
48
+ "language_model.model.layers.10.mlp.up_proj.weight": "model.safetensors",
49
+ "language_model.model.layers.10.post_attention_layernorm.weight": "model.safetensors",
50
+ "language_model.model.layers.11.input_layernorm.weight": "model.safetensors",
51
+ "language_model.model.layers.11.mlp.down_proj.weight": "model.safetensors",
52
+ "language_model.model.layers.11.mlp.gate_proj.weight": "model.safetensors",
53
+ "language_model.model.layers.11.mlp.up_proj.weight": "model.safetensors",
54
+ "language_model.model.layers.11.post_attention_layernorm.weight": "model.safetensors",
55
+ "language_model.model.layers.11.self_attn.k_norm.weight": "model.safetensors",
56
+ "language_model.model.layers.11.self_attn.k_proj.weight": "model.safetensors",
57
+ "language_model.model.layers.11.self_attn.o_proj.weight": "model.safetensors",
58
+ "language_model.model.layers.11.self_attn.q_norm.weight": "model.safetensors",
59
+ "language_model.model.layers.11.self_attn.q_proj.weight": "model.safetensors",
60
+ "language_model.model.layers.11.self_attn.v_proj.weight": "model.safetensors",
61
+ "language_model.model.layers.12.input_layernorm.weight": "model.safetensors",
62
+ "language_model.model.layers.12.linear_attn.A_log": "model.safetensors",
63
+ "language_model.model.layers.12.linear_attn.conv1d.weight": "model.safetensors",
64
+ "language_model.model.layers.12.linear_attn.dt_bias": "model.safetensors",
65
+ "language_model.model.layers.12.linear_attn.in_proj_a.weight": "model.safetensors",
66
+ "language_model.model.layers.12.linear_attn.in_proj_b.weight": "model.safetensors",
67
+ "language_model.model.layers.12.linear_attn.in_proj_qkv.weight": "model.safetensors",
68
+ "language_model.model.layers.12.linear_attn.in_proj_z.weight": "model.safetensors",
69
+ "language_model.model.layers.12.linear_attn.norm.weight": "model.safetensors",
70
+ "language_model.model.layers.12.linear_attn.out_proj.weight": "model.safetensors",
71
+ "language_model.model.layers.12.mlp.down_proj.weight": "model.safetensors",
72
+ "language_model.model.layers.12.mlp.gate_proj.weight": "model.safetensors",
73
+ "language_model.model.layers.12.mlp.up_proj.weight": "model.safetensors",
74
+ "language_model.model.layers.12.post_attention_layernorm.weight": "model.safetensors",
75
+ "language_model.model.layers.13.input_layernorm.weight": "model.safetensors",
76
+ "language_model.model.layers.13.linear_attn.A_log": "model.safetensors",
77
+ "language_model.model.layers.13.linear_attn.conv1d.weight": "model.safetensors",
78
+ "language_model.model.layers.13.linear_attn.dt_bias": "model.safetensors",
79
+ "language_model.model.layers.13.linear_attn.in_proj_a.weight": "model.safetensors",
80
+ "language_model.model.layers.13.linear_attn.in_proj_b.weight": "model.safetensors",
81
+ "language_model.model.layers.13.linear_attn.in_proj_qkv.weight": "model.safetensors",
82
+ "language_model.model.layers.13.linear_attn.in_proj_z.weight": "model.safetensors",
83
+ "language_model.model.layers.13.linear_attn.norm.weight": "model.safetensors",
84
+ "language_model.model.layers.13.linear_attn.out_proj.weight": "model.safetensors",
85
+ "language_model.model.layers.13.mlp.down_proj.weight": "model.safetensors",
86
+ "language_model.model.layers.13.mlp.gate_proj.weight": "model.safetensors",
87
+ "language_model.model.layers.13.mlp.up_proj.weight": "model.safetensors",
88
+ "language_model.model.layers.13.post_attention_layernorm.weight": "model.safetensors",
89
+ "language_model.model.layers.14.input_layernorm.weight": "model.safetensors",
90
+ "language_model.model.layers.14.linear_attn.A_log": "model.safetensors",
91
+ "language_model.model.layers.14.linear_attn.conv1d.weight": "model.safetensors",
92
+ "language_model.model.layers.14.linear_attn.dt_bias": "model.safetensors",
93
+ "language_model.model.layers.14.linear_attn.in_proj_a.weight": "model.safetensors",
94
+ "language_model.model.layers.14.linear_attn.in_proj_b.weight": "model.safetensors",
95
+ "language_model.model.layers.14.linear_attn.in_proj_qkv.weight": "model.safetensors",
96
+ "language_model.model.layers.14.linear_attn.in_proj_z.weight": "model.safetensors",
97
+ "language_model.model.layers.14.linear_attn.norm.weight": "model.safetensors",
98
+ "language_model.model.layers.14.linear_attn.out_proj.weight": "model.safetensors",
99
+ "language_model.model.layers.14.mlp.down_proj.weight": "model.safetensors",
100
+ "language_model.model.layers.14.mlp.gate_proj.weight": "model.safetensors",
101
+ "language_model.model.layers.14.mlp.up_proj.weight": "model.safetensors",
102
+ "language_model.model.layers.14.post_attention_layernorm.weight": "model.safetensors",
103
+ "language_model.model.layers.15.input_layernorm.weight": "model.safetensors",
104
+ "language_model.model.layers.15.mlp.down_proj.weight": "model.safetensors",
105
+ "language_model.model.layers.15.mlp.gate_proj.weight": "model.safetensors",
106
+ "language_model.model.layers.15.mlp.up_proj.weight": "model.safetensors",
107
+ "language_model.model.layers.15.post_attention_layernorm.weight": "model.safetensors",
108
+ "language_model.model.layers.15.self_attn.k_norm.weight": "model.safetensors",
109
+ "language_model.model.layers.15.self_attn.k_proj.weight": "model.safetensors",
110
+ "language_model.model.layers.15.self_attn.o_proj.weight": "model.safetensors",
111
+ "language_model.model.layers.15.self_attn.q_norm.weight": "model.safetensors",
112
+ "language_model.model.layers.15.self_attn.q_proj.weight": "model.safetensors",
113
+ "language_model.model.layers.15.self_attn.v_proj.weight": "model.safetensors",
114
+ "language_model.model.layers.16.input_layernorm.weight": "model.safetensors",
115
+ "language_model.model.layers.16.linear_attn.A_log": "model.safetensors",
116
+ "language_model.model.layers.16.linear_attn.conv1d.weight": "model.safetensors",
117
+ "language_model.model.layers.16.linear_attn.dt_bias": "model.safetensors",
118
+ "language_model.model.layers.16.linear_attn.in_proj_a.weight": "model.safetensors",
119
+ "language_model.model.layers.16.linear_attn.in_proj_b.weight": "model.safetensors",
120
+ "language_model.model.layers.16.linear_attn.in_proj_qkv.weight": "model.safetensors",
121
+ "language_model.model.layers.16.linear_attn.in_proj_z.weight": "model.safetensors",
122
+ "language_model.model.layers.16.linear_attn.norm.weight": "model.safetensors",
123
+ "language_model.model.layers.16.linear_attn.out_proj.weight": "model.safetensors",
124
+ "language_model.model.layers.16.mlp.down_proj.weight": "model.safetensors",
125
+ "language_model.model.layers.16.mlp.gate_proj.weight": "model.safetensors",
126
+ "language_model.model.layers.16.mlp.up_proj.weight": "model.safetensors",
127
+ "language_model.model.layers.16.post_attention_layernorm.weight": "model.safetensors",
128
+ "language_model.model.layers.17.input_layernorm.weight": "model.safetensors",
129
+ "language_model.model.layers.17.linear_attn.A_log": "model.safetensors",
130
+ "language_model.model.layers.17.linear_attn.conv1d.weight": "model.safetensors",
131
+ "language_model.model.layers.17.linear_attn.dt_bias": "model.safetensors",
132
+ "language_model.model.layers.17.linear_attn.in_proj_a.weight": "model.safetensors",
133
+ "language_model.model.layers.17.linear_attn.in_proj_b.weight": "model.safetensors",
134
+ "language_model.model.layers.17.linear_attn.in_proj_qkv.weight": "model.safetensors",
135
+ "language_model.model.layers.17.linear_attn.in_proj_z.weight": "model.safetensors",
136
+ "language_model.model.layers.17.linear_attn.norm.weight": "model.safetensors",
137
+ "language_model.model.layers.17.linear_attn.out_proj.weight": "model.safetensors",
138
+ "language_model.model.layers.17.mlp.down_proj.weight": "model.safetensors",
139
+ "language_model.model.layers.17.mlp.gate_proj.weight": "model.safetensors",
140
+ "language_model.model.layers.17.mlp.up_proj.weight": "model.safetensors",
141
+ "language_model.model.layers.17.post_attention_layernorm.weight": "model.safetensors",
142
+ "language_model.model.layers.18.input_layernorm.weight": "model.safetensors",
143
+ "language_model.model.layers.18.linear_attn.A_log": "model.safetensors",
144
+ "language_model.model.layers.18.linear_attn.conv1d.weight": "model.safetensors",
145
+ "language_model.model.layers.18.linear_attn.dt_bias": "model.safetensors",
146
+ "language_model.model.layers.18.linear_attn.in_proj_a.weight": "model.safetensors",
147
+ "language_model.model.layers.18.linear_attn.in_proj_b.weight": "model.safetensors",
148
+ "language_model.model.layers.18.linear_attn.in_proj_qkv.weight": "model.safetensors",
149
+ "language_model.model.layers.18.linear_attn.in_proj_z.weight": "model.safetensors",
150
+ "language_model.model.layers.18.linear_attn.norm.weight": "model.safetensors",
151
+ "language_model.model.layers.18.linear_attn.out_proj.weight": "model.safetensors",
152
+ "language_model.model.layers.18.mlp.down_proj.weight": "model.safetensors",
153
+ "language_model.model.layers.18.mlp.gate_proj.weight": "model.safetensors",
154
+ "language_model.model.layers.18.mlp.up_proj.weight": "model.safetensors",
155
+ "language_model.model.layers.18.post_attention_layernorm.weight": "model.safetensors",
156
+ "language_model.model.layers.19.input_layernorm.weight": "model.safetensors",
157
+ "language_model.model.layers.19.mlp.down_proj.weight": "model.safetensors",
158
+ "language_model.model.layers.19.mlp.gate_proj.weight": "model.safetensors",
159
+ "language_model.model.layers.19.mlp.up_proj.weight": "model.safetensors",
160
+ "language_model.model.layers.19.post_attention_layernorm.weight": "model.safetensors",
161
+ "language_model.model.layers.19.self_attn.k_norm.weight": "model.safetensors",
162
+ "language_model.model.layers.19.self_attn.k_proj.weight": "model.safetensors",
163
+ "language_model.model.layers.19.self_attn.o_proj.weight": "model.safetensors",
164
+ "language_model.model.layers.19.self_attn.q_norm.weight": "model.safetensors",
165
+ "language_model.model.layers.19.self_attn.q_proj.weight": "model.safetensors",
166
+ "language_model.model.layers.19.self_attn.v_proj.weight": "model.safetensors",
167
+ "language_model.model.layers.2.input_layernorm.weight": "model.safetensors",
168
+ "language_model.model.layers.2.linear_attn.A_log": "model.safetensors",
169
+ "language_model.model.layers.2.linear_attn.conv1d.weight": "model.safetensors",
170
+ "language_model.model.layers.2.linear_attn.dt_bias": "model.safetensors",
171
+ "language_model.model.layers.2.linear_attn.in_proj_a.weight": "model.safetensors",
172
+ "language_model.model.layers.2.linear_attn.in_proj_b.weight": "model.safetensors",
173
+ "language_model.model.layers.2.linear_attn.in_proj_qkv.weight": "model.safetensors",
174
+ "language_model.model.layers.2.linear_attn.in_proj_z.weight": "model.safetensors",
175
+ "language_model.model.layers.2.linear_attn.norm.weight": "model.safetensors",
176
+ "language_model.model.layers.2.linear_attn.out_proj.weight": "model.safetensors",
177
+ "language_model.model.layers.2.mlp.down_proj.weight": "model.safetensors",
178
+ "language_model.model.layers.2.mlp.gate_proj.weight": "model.safetensors",
179
+ "language_model.model.layers.2.mlp.up_proj.weight": "model.safetensors",
180
+ "language_model.model.layers.2.post_attention_layernorm.weight": "model.safetensors",
181
+ "language_model.model.layers.20.input_layernorm.weight": "model.safetensors",
182
+ "language_model.model.layers.20.linear_attn.A_log": "model.safetensors",
183
+ "language_model.model.layers.20.linear_attn.conv1d.weight": "model.safetensors",
184
+ "language_model.model.layers.20.linear_attn.dt_bias": "model.safetensors",
185
+ "language_model.model.layers.20.linear_attn.in_proj_a.weight": "model.safetensors",
186
+ "language_model.model.layers.20.linear_attn.in_proj_b.weight": "model.safetensors",
187
+ "language_model.model.layers.20.linear_attn.in_proj_qkv.weight": "model.safetensors",
188
+ "language_model.model.layers.20.linear_attn.in_proj_z.weight": "model.safetensors",
189
+ "language_model.model.layers.20.linear_attn.norm.weight": "model.safetensors",
190
+ "language_model.model.layers.20.linear_attn.out_proj.weight": "model.safetensors",
191
+ "language_model.model.layers.20.mlp.down_proj.weight": "model.safetensors",
192
+ "language_model.model.layers.20.mlp.gate_proj.weight": "model.safetensors",
193
+ "language_model.model.layers.20.mlp.up_proj.weight": "model.safetensors",
194
+ "language_model.model.layers.20.post_attention_layernorm.weight": "model.safetensors",
195
+ "language_model.model.layers.21.input_layernorm.weight": "model.safetensors",
196
+ "language_model.model.layers.21.linear_attn.A_log": "model.safetensors",
197
+ "language_model.model.layers.21.linear_attn.conv1d.weight": "model.safetensors",
198
+ "language_model.model.layers.21.linear_attn.dt_bias": "model.safetensors",
199
+ "language_model.model.layers.21.linear_attn.in_proj_a.weight": "model.safetensors",
200
+ "language_model.model.layers.21.linear_attn.in_proj_b.weight": "model.safetensors",
201
+ "language_model.model.layers.21.linear_attn.in_proj_qkv.weight": "model.safetensors",
202
+ "language_model.model.layers.21.linear_attn.in_proj_z.weight": "model.safetensors",
203
+ "language_model.model.layers.21.linear_attn.norm.weight": "model.safetensors",
204
+ "language_model.model.layers.21.linear_attn.out_proj.weight": "model.safetensors",
205
+ "language_model.model.layers.21.mlp.down_proj.weight": "model.safetensors",
206
+ "language_model.model.layers.21.mlp.gate_proj.weight": "model.safetensors",
207
+ "language_model.model.layers.21.mlp.up_proj.weight": "model.safetensors",
208
+ "language_model.model.layers.21.post_attention_layernorm.weight": "model.safetensors",
209
+ "language_model.model.layers.22.input_layernorm.weight": "model.safetensors",
210
+ "language_model.model.layers.22.linear_attn.A_log": "model.safetensors",
211
+ "language_model.model.layers.22.linear_attn.conv1d.weight": "model.safetensors",
212
+ "language_model.model.layers.22.linear_attn.dt_bias": "model.safetensors",
213
+ "language_model.model.layers.22.linear_attn.in_proj_a.weight": "model.safetensors",
214
+ "language_model.model.layers.22.linear_attn.in_proj_b.weight": "model.safetensors",
215
+ "language_model.model.layers.22.linear_attn.in_proj_qkv.weight": "model.safetensors",
216
+ "language_model.model.layers.22.linear_attn.in_proj_z.weight": "model.safetensors",
217
+ "language_model.model.layers.22.linear_attn.norm.weight": "model.safetensors",
218
+ "language_model.model.layers.22.linear_attn.out_proj.weight": "model.safetensors",
219
+ "language_model.model.layers.22.mlp.down_proj.weight": "model.safetensors",
220
+ "language_model.model.layers.22.mlp.gate_proj.weight": "model.safetensors",
221
+ "language_model.model.layers.22.mlp.up_proj.weight": "model.safetensors",
222
+ "language_model.model.layers.22.post_attention_layernorm.weight": "model.safetensors",
223
+ "language_model.model.layers.23.input_layernorm.weight": "model.safetensors",
224
+ "language_model.model.layers.23.mlp.down_proj.weight": "model.safetensors",
225
+ "language_model.model.layers.23.mlp.gate_proj.weight": "model.safetensors",
226
+ "language_model.model.layers.23.mlp.up_proj.weight": "model.safetensors",
227
+ "language_model.model.layers.23.post_attention_layernorm.weight": "model.safetensors",
228
+ "language_model.model.layers.23.self_attn.k_norm.weight": "model.safetensors",
229
+ "language_model.model.layers.23.self_attn.k_proj.weight": "model.safetensors",
230
+ "language_model.model.layers.23.self_attn.o_proj.weight": "model.safetensors",
231
+ "language_model.model.layers.23.self_attn.q_norm.weight": "model.safetensors",
232
+ "language_model.model.layers.23.self_attn.q_proj.weight": "model.safetensors",
233
+ "language_model.model.layers.23.self_attn.v_proj.weight": "model.safetensors",
234
+ "language_model.model.layers.3.input_layernorm.weight": "model.safetensors",
235
+ "language_model.model.layers.3.mlp.down_proj.weight": "model.safetensors",
236
+ "language_model.model.layers.3.mlp.gate_proj.weight": "model.safetensors",
237
+ "language_model.model.layers.3.mlp.up_proj.weight": "model.safetensors",
238
+ "language_model.model.layers.3.post_attention_layernorm.weight": "model.safetensors",
239
+ "language_model.model.layers.3.self_attn.k_norm.weight": "model.safetensors",
240
+ "language_model.model.layers.3.self_attn.k_proj.weight": "model.safetensors",
241
+ "language_model.model.layers.3.self_attn.o_proj.weight": "model.safetensors",
242
+ "language_model.model.layers.3.self_attn.q_norm.weight": "model.safetensors",
243
+ "language_model.model.layers.3.self_attn.q_proj.weight": "model.safetensors",
244
+ "language_model.model.layers.3.self_attn.v_proj.weight": "model.safetensors",
245
+ "language_model.model.layers.4.input_layernorm.weight": "model.safetensors",
246
+ "language_model.model.layers.4.linear_attn.A_log": "model.safetensors",
247
+ "language_model.model.layers.4.linear_attn.conv1d.weight": "model.safetensors",
248
+ "language_model.model.layers.4.linear_attn.dt_bias": "model.safetensors",
249
+ "language_model.model.layers.4.linear_attn.in_proj_a.weight": "model.safetensors",
250
+ "language_model.model.layers.4.linear_attn.in_proj_b.weight": "model.safetensors",
251
+ "language_model.model.layers.4.linear_attn.in_proj_qkv.weight": "model.safetensors",
252
+ "language_model.model.layers.4.linear_attn.in_proj_z.weight": "model.safetensors",
253
+ "language_model.model.layers.4.linear_attn.norm.weight": "model.safetensors",
254
+ "language_model.model.layers.4.linear_attn.out_proj.weight": "model.safetensors",
255
+ "language_model.model.layers.4.mlp.down_proj.weight": "model.safetensors",
256
+ "language_model.model.layers.4.mlp.gate_proj.weight": "model.safetensors",
257
+ "language_model.model.layers.4.mlp.up_proj.weight": "model.safetensors",
258
+ "language_model.model.layers.4.post_attention_layernorm.weight": "model.safetensors",
259
+ "language_model.model.layers.5.input_layernorm.weight": "model.safetensors",
260
+ "language_model.model.layers.5.linear_attn.A_log": "model.safetensors",
261
+ "language_model.model.layers.5.linear_attn.conv1d.weight": "model.safetensors",
262
+ "language_model.model.layers.5.linear_attn.dt_bias": "model.safetensors",
263
+ "language_model.model.layers.5.linear_attn.in_proj_a.weight": "model.safetensors",
264
+ "language_model.model.layers.5.linear_attn.in_proj_b.weight": "model.safetensors",
265
+ "language_model.model.layers.5.linear_attn.in_proj_qkv.weight": "model.safetensors",
266
+ "language_model.model.layers.5.linear_attn.in_proj_z.weight": "model.safetensors",
267
+ "language_model.model.layers.5.linear_attn.norm.weight": "model.safetensors",
268
+ "language_model.model.layers.5.linear_attn.out_proj.weight": "model.safetensors",
269
+ "language_model.model.layers.5.mlp.down_proj.weight": "model.safetensors",
270
+ "language_model.model.layers.5.mlp.gate_proj.weight": "model.safetensors",
271
+ "language_model.model.layers.5.mlp.up_proj.weight": "model.safetensors",
272
+ "language_model.model.layers.5.post_attention_layernorm.weight": "model.safetensors",
273
+ "language_model.model.layers.6.input_layernorm.weight": "model.safetensors",
274
+ "language_model.model.layers.6.linear_attn.A_log": "model.safetensors",
275
+ "language_model.model.layers.6.linear_attn.conv1d.weight": "model.safetensors",
276
+ "language_model.model.layers.6.linear_attn.dt_bias": "model.safetensors",
277
+ "language_model.model.layers.6.linear_attn.in_proj_a.weight": "model.safetensors",
278
+ "language_model.model.layers.6.linear_attn.in_proj_b.weight": "model.safetensors",
279
+ "language_model.model.layers.6.linear_attn.in_proj_qkv.weight": "model.safetensors",
280
+ "language_model.model.layers.6.linear_attn.in_proj_z.weight": "model.safetensors",
281
+ "language_model.model.layers.6.linear_attn.norm.weight": "model.safetensors",
282
+ "language_model.model.layers.6.linear_attn.out_proj.weight": "model.safetensors",
283
+ "language_model.model.layers.6.mlp.down_proj.weight": "model.safetensors",
284
+ "language_model.model.layers.6.mlp.gate_proj.weight": "model.safetensors",
285
+ "language_model.model.layers.6.mlp.up_proj.weight": "model.safetensors",
286
+ "language_model.model.layers.6.post_attention_layernorm.weight": "model.safetensors",
287
+ "language_model.model.layers.7.input_layernorm.weight": "model.safetensors",
288
+ "language_model.model.layers.7.mlp.down_proj.weight": "model.safetensors",
289
+ "language_model.model.layers.7.mlp.gate_proj.weight": "model.safetensors",
290
+ "language_model.model.layers.7.mlp.up_proj.weight": "model.safetensors",
291
+ "language_model.model.layers.7.post_attention_layernorm.weight": "model.safetensors",
292
+ "language_model.model.layers.7.self_attn.k_norm.weight": "model.safetensors",
293
+ "language_model.model.layers.7.self_attn.k_proj.weight": "model.safetensors",
294
+ "language_model.model.layers.7.self_attn.o_proj.weight": "model.safetensors",
295
+ "language_model.model.layers.7.self_attn.q_norm.weight": "model.safetensors",
296
+ "language_model.model.layers.7.self_attn.q_proj.weight": "model.safetensors",
297
+ "language_model.model.layers.7.self_attn.v_proj.weight": "model.safetensors",
298
+ "language_model.model.layers.8.input_layernorm.weight": "model.safetensors",
299
+ "language_model.model.layers.8.linear_attn.A_log": "model.safetensors",
300
+ "language_model.model.layers.8.linear_attn.conv1d.weight": "model.safetensors",
301
+ "language_model.model.layers.8.linear_attn.dt_bias": "model.safetensors",
302
+ "language_model.model.layers.8.linear_attn.in_proj_a.weight": "model.safetensors",
303
+ "language_model.model.layers.8.linear_attn.in_proj_b.weight": "model.safetensors",
304
+ "language_model.model.layers.8.linear_attn.in_proj_qkv.weight": "model.safetensors",
305
+ "language_model.model.layers.8.linear_attn.in_proj_z.weight": "model.safetensors",
306
+ "language_model.model.layers.8.linear_attn.norm.weight": "model.safetensors",
307
+ "language_model.model.layers.8.linear_attn.out_proj.weight": "model.safetensors",
308
+ "language_model.model.layers.8.mlp.down_proj.weight": "model.safetensors",
309
+ "language_model.model.layers.8.mlp.gate_proj.weight": "model.safetensors",
310
+ "language_model.model.layers.8.mlp.up_proj.weight": "model.safetensors",
311
+ "language_model.model.layers.8.post_attention_layernorm.weight": "model.safetensors",
312
+ "language_model.model.layers.9.input_layernorm.weight": "model.safetensors",
313
+ "language_model.model.layers.9.linear_attn.A_log": "model.safetensors",
314
+ "language_model.model.layers.9.linear_attn.conv1d.weight": "model.safetensors",
315
+ "language_model.model.layers.9.linear_attn.dt_bias": "model.safetensors",
316
+ "language_model.model.layers.9.linear_attn.in_proj_a.weight": "model.safetensors",
317
+ "language_model.model.layers.9.linear_attn.in_proj_b.weight": "model.safetensors",
318
+ "language_model.model.layers.9.linear_attn.in_proj_qkv.weight": "model.safetensors",
319
+ "language_model.model.layers.9.linear_attn.in_proj_z.weight": "model.safetensors",
320
+ "language_model.model.layers.9.linear_attn.norm.weight": "model.safetensors",
321
+ "language_model.model.layers.9.linear_attn.out_proj.weight": "model.safetensors",
322
+ "language_model.model.layers.9.mlp.down_proj.weight": "model.safetensors",
323
+ "language_model.model.layers.9.mlp.gate_proj.weight": "model.safetensors",
324
+ "language_model.model.layers.9.mlp.up_proj.weight": "model.safetensors",
325
+ "language_model.model.layers.9.post_attention_layernorm.weight": "model.safetensors",
326
+ "language_model.model.norm.weight": "model.safetensors"
327
+ }
328
+ }
bf16/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
bf16/tokenizer_config.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "tool_parser_type": "qwen3_coder",
29
+ "unk_token": null,
30
+ "video_token": "<|video_pad|>",
31
+ "vision_bos_token": "<|vision_start|>",
32
+ "vision_eos_token": "<|vision_end|>"
33
+ }
config.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 2048,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 6144,
17
+ "layer_types": [
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "linear_attention",
21
+ "full_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention"
42
+ ],
43
+ "linear_conv_kernel_dim": 4,
44
+ "linear_key_head_dim": 128,
45
+ "linear_num_key_heads": 16,
46
+ "linear_num_value_heads": 16,
47
+ "linear_value_head_dim": 128,
48
+ "mamba_ssm_dtype": "float32",
49
+ "max_position_embeddings": 262144,
50
+ "mlp_only_layers": [],
51
+ "model_type": "qwen3_5",
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_hidden_layers": 24,
56
+ "num_key_value_heads": 2,
57
+ "pad_token_id": null,
58
+ "partial_rotary_factor": 0.25,
59
+ "rms_norm_eps": 1e-06,
60
+ "rope_parameters": {
61
+ "mrope_interleaved": true,
62
+ "mrope_section": [
63
+ 11,
64
+ 11,
65
+ 10
66
+ ],
67
+ "partial_rotary_factor": 0.25,
68
+ "rope_theta": 10000000,
69
+ "type": "default"
70
+ },
71
+ "tie_word_embeddings": true,
72
+ "transformers_version": "5.17.0",
73
+ "use_cache": false,
74
+ "vocab_size": 248320
75
+ }
figures/banner.data.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "left": {
3
+ "big": "73.6% on JevBench",
4
+ "sub": "+9.5 points over our 0.8B v3; Jev 86.6%",
5
+ "tokens": "25,600 tokens per call, no option cap"
6
+ },
7
+ "rows": [
8
+ {
9
+ "label": "Jev 1.13 (API)",
10
+ "accuracy": 0.8658008658008658,
11
+ "pct_1dp": 86.6
12
+ },
13
+ {
14
+ "label": "2B v3 \u00b7 this model",
15
+ "accuracy": 0.7359307359307359,
16
+ "pct_1dp": 73.6
17
+ },
18
+ {
19
+ "label": "decider-2b",
20
+ "accuracy": 0.70995670995671,
21
+ "pct_1dp": 71.0
22
+ },
23
+ {
24
+ "label": "Open-Jev 2B",
25
+ "accuracy": 0.645021645021645,
26
+ "pct_1dp": 64.5
27
+ },
28
+ {
29
+ "label": "0.8B v3",
30
+ "accuracy": 0.6406926406926406,
31
+ "pct_1dp": 64.1
32
+ },
33
+ {
34
+ "label": "Laya",
35
+ "accuracy": 0.5844155844155844,
36
+ "pct_1dp": 58.4
37
+ }
38
+ ],
39
+ "shown": "Shown: Qwen3.5-2B-family systems, our 0.8B v3, Laya and Jev \u00b7 42 of 82 board systems score higher than 73.6%",
40
+ "note": "231 public items. 2B v3: self-run with the official harness (GGUF F16), not an official board entry; 95% CI 67.6\u201378.9%, so the lead over decider-2b is inside the CI. Other rows as published on the v1.4.1 board.",
41
+ "sources": [
42
+ "figures/jevbench.data.json"
43
+ ]
44
+ }
figures/banner.png ADDED

Git LFS Details

  • SHA256: 2623e4afad36239aaa007c34350d6f2fbacb33430ffd3c36ce2f7ee7db849b47
  • Pointer size: 131 Bytes
  • Size of remote file: 447 kB
figures/jevbench.data.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "jevbench",
3
+ "metric": "JevBench v1.4.1 public accuracy (231 items)",
4
+ "rows": [
5
+ {
6
+ "label": "Jev-Style 2B v3 (this model)",
7
+ "accuracy": 0.7359307359307359,
8
+ "accuracy_pct_1dp": 73.6,
9
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/jevbench_v1.4.1_results.json :: public_accuracy (170/231)"
10
+ },
11
+ {
12
+ "label": "decider-2b (Mapika, Qwen3.5-2B-Base)",
13
+ "accuracy": 0.70995670995671,
14
+ "accuracy_pct_1dp": 71.0,
15
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[decider-2b]"
16
+ },
17
+ {
18
+ "label": "Open-Jev 2B (Zefan Cai, Qwen3.5-2B + LoRA)",
19
+ "accuracy": 0.645021645021645,
20
+ "accuracy_pct_1dp": 64.5,
21
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[open-jev-zefan-2b]"
22
+ },
23
+ {
24
+ "label": "Jev-Style 0.8B v3",
25
+ "accuracy": 0.6406926406926406,
26
+ "accuracy_pct_1dp": 64.1,
27
+ "source": "https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3/blob/main/figures/jevbench.data.json (0.8B v3 card; row Jev-Style 0.8B v3, 148/231)"
28
+ },
29
+ {
30
+ "label": "Laya (ModernBERT-large, 421M)",
31
+ "accuracy": 0.5844155844155844,
32
+ "accuracy_pct_1dp": 58.4,
33
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[laya]"
34
+ }
35
+ ],
36
+ "reference_line": {
37
+ "label": "Jev 1.13.0",
38
+ "accuracy": 0.8658008658008658,
39
+ "source": "https://github.com/fstandhartinger/jevbench/blob/v1.4.1/results/v1.4.1/jevbench-v1.4.1-results.json :: systems[jev-1.13.0]"
40
+ },
41
+ "ci95_wilson_2b": [
42
+ 0.6755578394149232,
43
+ 0.7885850787366094
44
+ ],
45
+ "board_systems_higher_than_2b": 42,
46
+ "board_systems": 82,
47
+ "board_sha256": "e6754863056503fe2b010410fc7111df884ac1f9ce4449aa369aab61d98092cd",
48
+ "footnote": "JevBench v1.4.1, 231 public items. 2B v3: self-run once with the official harness (commit 24b9b5c), GGUF F16 engine, one global temperature, not an official board entry; 95% CI 67.6-78.9% (Wilson), so its lead over decider-2b (164/231) is inside the CI; training-pool contamination scan: 0 hits. Other rows: public accuracy as published in the board's v1.4.1 results file. Shown: the Qwen3.5-2B-family systems on the board, our 0.8B v3, Laya and Jev; 42 of the 82 board systems score higher than 73.6%."
49
+ }
figures/jevbench.png ADDED

Git LFS Details

  • SHA256: c8542dba9ecc1bd2a4ea3bda844cbc3fbd661c40d6eb9254098b279a92a74ce8
  • Pointer size: 131 Bytes
  • Size of remote file: 168 kB
figures/jevbench.svg ADDED
figures/zeroshot.data.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chart": "zeroshot",
3
+ "metric": "accuracy over every row of the pinned test files",
4
+ "values": {
5
+ "tweet_topic": {
6
+ "2b": 0.822209096278795,
7
+ "2b_ci95": [
8
+ 0.8038984051978736,
9
+ 0.8399438865918486
10
+ ],
11
+ "2b_macro_f1": 0.677897069437572,
12
+ "2b_ece15": 0.027919883880813873,
13
+ "08b": 0.754873006497342,
14
+ "jev": 0.7932663910218547,
15
+ "jev_macro_f1": 0.6936,
16
+ "jev_ece15": 0.0631,
17
+ "n": 1693
18
+ },
19
+ "fin_topic": {
20
+ "2b": 0.6111246052951178,
21
+ "2b_ci95": [
22
+ 0.5960650959436483,
23
+ 0.6259412193344669
24
+ ],
25
+ "2b_macro_f1": 0.5897964463354406,
26
+ "2b_ece15": 0.06460657860002232,
27
+ "08b": 0.4670876852076755,
28
+ "jev": 0.669905270828273,
29
+ "jev_macro_f1": 0.6298,
30
+ "jev_ece15": 0.1664,
31
+ "n": 4117
32
+ }
33
+ },
34
+ "sources": {
35
+ "2b": "https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/zeroshot_metrics.json",
36
+ "0.8b": "https://huggingface.co/chaoliangUNSW/Jev-Style-0.8B-Decision-v3/blob/main/figures/zeroshot.json (0.8B v3 card; v3_recomputed: tweet_topic 1278/1693, fin_topic 1923/4117)",
37
+ "jev": "https://github.com/elcronos/jev-vs-open-decision-models/blob/a1901bc3d520e73936de8d4326545c0cdcf742fb/results/cross_dataset_summary.json (as copied in https://huggingface.co/chaoliangUNSW/Jev-Style-2B-Decision-v3/blob/main/validation/benchmarks/zeroshot_metrics.json :: comparison)"
38
+ },
39
+ "footnote": "Zero-shot: none of these test sets is in the 2B or 0.8B training pool; accuracy over every row of the pinned test files (n = 1,693 and 4,117). 2B v3: GGUF F16 engine, one global temperature, run once; tweet_topic 95% CI 80.4-84.0%. Jev (1.13, API): numbers published by the elcronos jev-vs-open-decision-models study (cross_dataset_summary.json @ a1901bc), not re-run by us. Macro-F1 is below Jev on both sets (tweet_topic 67.8% vs 69.4%; fin_topic 59.0% vs 63.0%)."
40
+ }
figures/zeroshot.png ADDED

Git LFS Details

  • SHA256: d51f9a7bc46b88d8dcf5ba46f15d479c781302ca531208ff0838106da8aae755
  • Pointer size: 131 Bytes
  • Size of remote file: 138 kB
figures/zeroshot.svg ADDED
jev_style_decision_mlx.py ADDED
@@ -0,0 +1,1059 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Jev-Style-2B-Decision-v3: typed decisions with MLX on Apple silicon (bf16 / 8-bit, --precision).
2
+
3
+ Self-contained runtime for chaoliangUNSW/Jev-Style-2B-Decision-v3 (Apache-2.0). No dependency on any training
4
+ code: rendering (protocol "macjev-render-v2-long-options"), block-causal attention, the verdict readout and
5
+ calibration are implemented below and reproduce the reference implementation used for evaluation (see
6
+ release_config.json -> "runtime_parity"). Needs mlx, mlx-lm==0.31.3 (pinned, see the MLX backend section),
7
+ tokenizers and numpy.
8
+
9
+ Calibration: probabilities use the ONE global temperature of readout_config.json (temperatures.global);
10
+ temperature=... (CLI --temperature, JSONL "temperature") overrides it (1.0 = uncalibrated scores). Unlike the
11
+ 0.8B v3 runtime there is no --category (no group temperatures) and no --head-max (no question/options cap:
12
+ only the 25,600-token total budget applies).
13
+
14
+ Python:
15
+ from jev_style_decision_mlx import JevStyleDecisionMLX
16
+ m = JevStyleDecisionMLX(".", precision="bf16") # or "8bit"
17
+ m.decide("<state>", "Which action?", options={"click": "press it", "wait": None})
18
+ m.score_many("<state>", [q1, q2, ...]) # state computed once, reused for every question
19
+ """
20
+ # ----------------------------------------------------------------------------------------------
21
+ # Shared core (byte-identical in jev_style_decision.py, jev_style_decision_gguf.py and jev_style_decision_mlx.py):
22
+ # input rendering (v2), verdict readout, calibration, budgets, errors, manifest check, CLI/JSONL.
23
+ #
24
+ # Input layout ("macjev-render-v2-long-options", layout "sb"; token segments are encoded separately
25
+ # and concatenated, so slot positions are exact):
26
+ #
27
+ # state prefix State:\n<state>\n\n
28
+ # short form Question [<type>]: <question>\nOptions:\n
29
+ # (question + - <option 1> ->\n ... - <option K> ->\n
30
+ # options + slots <= 2,048 tokens)
31
+ # overflow form Question [<type>]: <question>\nOptions:\n
32
+ # (otherwise) Option 1: <option 1>\n ... Option K: <option K>\n (the catalogue)
33
+ # Judge each numbered option in the complete catalogue above:\n
34
+ # Option 1 ->\n ... Option K ->\n (the rubric)
35
+ #
36
+ # Attention blocks ([start, stop) token ranges): the state is cut into consecutive 2,048-token
37
+ # blocks; the short form is one block; in the overflow form the catalogue and the rubric are each cut
38
+ # into 2,048-token blocks. In the 6 full-attention layers every block attends to all earlier tokens
39
+ # and to itself with NO causal mask inside the block (block-causal); the Gated-DeltaNet layers are
40
+ # ordinary recurrent layers. A backend must compute exactly these blocks (never merged, never re-split).
41
+ #
42
+ # Score of option k = logit(" yes") - logit(" no") at its " ->" slot = h_slot . (W_yes - W_no) in
43
+ # float32 (final normed hidden state, tied embedding rows). Probabilities = softmax(scores / T) in
44
+ # canonical option order. T = the ONE global temperature of readout_config.json
45
+ # (temperatures.global, fitted on 2,000 calibration rows) unless temperature=... overrides it
46
+ # (1.0 = uncalibrated scores). This model has no category / group temperatures and no separate
47
+ # question/options budget, so the 0.8B v3 runtime's --category and --head-max options do not exist here.
48
+ #
49
+ # Budget: the complete input (state + question + options + readout) <= 25,600 tokens. Larger inputs
50
+ # raise InputBudgetError; nothing is ever truncated. Text inside the state, question and options is
51
+ # tokenised with special tokens disabled, so e.g. "<|im_end|>" in user text can never act as a
52
+ # control token. A non-finite score (NaN / inf) raises NonFiniteScoreError: no probabilities are made.
53
+ # ----------------------------------------------------------------------------------------------
54
+ import argparse
55
+ import hashlib
56
+ import json
57
+ import math
58
+ import sys
59
+ from pathlib import Path
60
+
61
+ import numpy as np
62
+
63
+ MODEL_NAME = "Jev-Style-2B-Decision-v3"
64
+ TEMPLATE_VERSION = "macjev-render-v2-long-options"
65
+ READOUT_FORMAT = "macjev-readout-v2"
66
+ LAYOUT = "sb"
67
+ BLOCK = 2048 # attention block size (a processing unit, not a content limit)
68
+ CONTEXT_LIMIT = 25_600 # state + question + options + readout, all included
69
+ QTYPES = ("choice", "score", "noul")
70
+ JSONL_GROUP_MAX = 256 # consecutive JSONL rows with one state scored in one backend call
71
+ HERE = Path(__file__).resolve().parent
72
+
73
+
74
+ class InputBudgetError(ValueError):
75
+ """The rendered input exceeds the token budget. Nothing was truncated."""
76
+
77
+
78
+ class QuestionError(ValueError):
79
+ """The question/options are malformed."""
80
+
81
+
82
+ class NonFiniteScoreError(FloatingPointError):
83
+ """The model produced a non-finite decision score (NaN or inf). No probabilities are returned."""
84
+
85
+
86
+ # -- questions ------------------------------------------------------------------------------------
87
+ def option_names(question):
88
+ """Canonical option identifiers, in the order the probabilities are returned."""
89
+ if not isinstance(question, dict):
90
+ raise QuestionError("question must be a dict {'t', 'ins', 'crit'}")
91
+ t, crit = question.get("t"), question.get("crit")
92
+ if not isinstance(question.get("ins"), str) or not question["ins"].strip():
93
+ raise QuestionError("question text ('ins') must be a non-empty string")
94
+ if t == "choice":
95
+ if not isinstance(crit, dict) or not crit:
96
+ raise QuestionError("choice needs a non-empty dict {option name: description or None}")
97
+ return [str(k) for k in crit]
98
+ if t == "score":
99
+ if not isinstance(crit, list) or not 2 <= len(crit) <= 10:
100
+ raise QuestionError("score needs a list of 2..10 level descriptions")
101
+ return [str(i) for i in range(len(crit))]
102
+ if t == "noul":
103
+ if crit is not None and not isinstance(crit, dict):
104
+ raise QuestionError("noul criteria must be None or {'false': ..., 'true': ...}")
105
+ return ["false", "true"]
106
+ raise QuestionError(f"unknown question type {t!r} (expected one of {QTYPES})")
107
+
108
+
109
+ def make_question(question, options=None, qtype=None):
110
+ """Build a typed question.
111
+
112
+ * ``question`` already a dict {"t", "ins", "crit"}: validated and returned.
113
+ * ``qtype="choice"`` (default when ``options`` is given): ``options`` = {name: description or None}
114
+ or a list of names.
115
+ * ``qtype="score"``: ``options`` = list of 2..10 level descriptions (level 0 first).
116
+ * ``qtype="noul"`` (default when no options): a true/false statement; ``options`` may be
117
+ {"false": "...", "true": "..."} to describe the two outcomes.
118
+ """
119
+ if isinstance(question, dict):
120
+ q = dict(question)
121
+ else:
122
+ if qtype is None:
123
+ qtype = "choice" if options is not None else "noul"
124
+ if qtype == "choice":
125
+ if isinstance(options, (list, tuple)):
126
+ if len(set(map(str, options))) != len(options):
127
+ raise QuestionError("duplicate option names")
128
+ crit = {str(o): None for o in options}
129
+ else:
130
+ crit = options
131
+ elif qtype == "score":
132
+ crit = list(options) if options is not None else None
133
+ else:
134
+ crit = options
135
+ q = {"t": qtype, "ins": question, "crit": crit}
136
+ option_names(q)
137
+ return q
138
+
139
+
140
+ def serialize_state(state):
141
+ """Strings pass through unchanged; any other JSON value is serialised (ensure_ascii=False)."""
142
+ if isinstance(state, str):
143
+ return state
144
+ return json.dumps(state, ensure_ascii=False)
145
+
146
+
147
+ def _criterion(value):
148
+ if isinstance(value, str):
149
+ return value
150
+ return json.dumps(value, ensure_ascii=False, separators=(", ", ": "), default=str)
151
+
152
+
153
+ def render_options(question):
154
+ t, crit = question["t"], question.get("crit")
155
+ if t == "choice":
156
+ return [k if v is None or v == "" else f"{k}: {_criterion(v)}" for k, v in crit.items()]
157
+ if t == "score":
158
+ return [f"level {i}: {_criterion(c)}" for i, c in enumerate(crit)]
159
+ crit = crit or {}
160
+ false_c, true_c = crit.get("false"), crit.get("true")
161
+ return ["false: " + (_criterion(false_c) if false_c not in (None, "") else "no, the statement does not hold"),
162
+ "true: " + (_criterion(true_c) if true_c not in (None, "") else "yes, the statement holds")]
163
+
164
+
165
+ # -- tokenizer + renderer -----------------------------------------------------------------------
166
+ class TextEncoder:
167
+ """HF ``tokenizers`` tokenizer.json; no BOS/EOS, special tokens in text are split (never control tokens)."""
168
+
169
+ def __init__(self, tokenizer_json):
170
+ from tokenizers import Tokenizer
171
+ self.tk = Tokenizer.from_file(str(tokenizer_json))
172
+ self.tk.encode_special_tokens = True
173
+
174
+ def __call__(self, text):
175
+ return self.tk.encode(text, add_special_tokens=False).ids
176
+
177
+ def id_to_token(self, i):
178
+ return self.tk.id_to_token(int(i))
179
+
180
+ def vocab_size(self):
181
+ return self.tk.get_vocab_size(with_added_tokens=True)
182
+
183
+
184
+ def _blocks(start, stop):
185
+ return [(s, min(s + BLOCK, stop)) for s in range(start, stop, BLOCK)]
186
+
187
+
188
+ class Rendered:
189
+ """One rendered question: token ids, the verdict slots (one per option, canonical order), the
190
+ attention blocks ([start, stop) pairs tiling [0, len(ids))) and the state prefix length."""
191
+ __slots__ = ("ids", "prefix_len", "slots", "blocks", "names", "qtype", "catalogue_overflow")
192
+
193
+ def __init__(self, ids, prefix_len, slots, blocks, names, qtype, catalogue_overflow):
194
+ self.ids, self.prefix_len, self.slots, self.blocks = ids, prefix_len, slots, blocks
195
+ self.names, self.qtype, self.catalogue_overflow = names, qtype, catalogue_overflow
196
+
197
+ @property
198
+ def state_blocks(self):
199
+ return [b for b in self.blocks if b[1] <= self.prefix_len]
200
+
201
+ @property
202
+ def question_blocks(self):
203
+ return [b for b in self.blocks if b[0] >= self.prefix_len]
204
+
205
+ @property
206
+ def input_tokens(self):
207
+ return len(self.ids)
208
+
209
+ @property
210
+ def head_tokens(self):
211
+ """question + options + readout tokens (everything after the state prefix)"""
212
+ return len(self.ids) - self.prefix_len
213
+
214
+
215
+ class Renderer:
216
+ def __init__(self, encode, readout_cfg, max_len=CONTEXT_LIMIT):
217
+ want = {"format": READOUT_FORMAT, "template": TEMPLATE_VERSION, "layout": LAYOUT, "readout": "verdict",
218
+ "block_size": BLOCK, "total_context_limit": CONTEXT_LIMIT}
219
+ bad = {k: readout_cfg.get(k) for k, v in want.items() if readout_cfg.get(k) != v}
220
+ if bad:
221
+ raise ValueError(f"readout_config.json does not describe this runtime's protocol {want}; got {bad}")
222
+ if not 0 < int(max_len) <= CONTEXT_LIMIT:
223
+ raise ValueError(f"max_len must be in 1..{CONTEXT_LIMIT}")
224
+ self.enc, self.max_len = encode, int(max_len)
225
+ self.head_max = self.max_len # no separate question/options cap (field kept for the 0.8B / jev-style API)
226
+ st = readout_cfg["slot_tokens"]
227
+ self.yes, self.no, arrow = int(st["yes"]["id"]), int(st["no"]["id"]), int(st["verdict_slot"]["id"])
228
+ for text, want_id in ((" yes", self.yes), (" no", self.no), (" ->", arrow)):
229
+ got = self.enc(text)
230
+ if got != [want_id]:
231
+ raise ValueError(f"tokenizer mismatch: {text!r} -> {got}, readout_config expects [{want_id}]")
232
+ self.arrow = [arrow]
233
+ self.newline = self.enc("\n")
234
+ self.dash = self.enc("- ")
235
+ self.rubric_head = self.enc("Judge each numbered option in the complete catalogue above:\n")
236
+ self._numbered = {}
237
+
238
+ def _option_label(self, pos, catalogue):
239
+ key = (pos, catalogue)
240
+ if key not in self._numbered:
241
+ self._numbered[key] = self.enc(f"Option {pos + 1}: " if catalogue else f"Option {pos + 1}")
242
+ return self._numbered[key]
243
+
244
+ def prefix_ids(self, state):
245
+ return self.enc("State:\n") + self.enc(serialize_state(state)) + self.enc("\n\n")
246
+
247
+ def pieces(self, state, question, prefix=None):
248
+ """Tokenised segments of one question (``prefix``: already tokenised state, to tokenise it once)."""
249
+ names = option_names(question)
250
+ return {"prefix": self.prefix_ids(state) if prefix is None else list(prefix),
251
+ "head": self.enc(f"Question [{question['t']}]: {question['ins']}\nOptions:\n"),
252
+ "opts": [self.enc(o) for o in render_options(question)], "names": names, "qtype": question["t"]}
253
+
254
+ def assemble(self, pieces, max_len=None):
255
+ maximum = self.max_len if max_len is None else min(self.max_len, int(max_len))
256
+ prefix, head, opts = list(pieces["prefix"]), list(pieces["head"]), pieces["opts"]
257
+ k = len(opts)
258
+ if k < 1:
259
+ raise QuestionError("at least one option is required")
260
+ short, rel = list(head), []
261
+ for o in opts:
262
+ short += self.dash + list(o) + self.arrow
263
+ rel.append(len(short) - 1)
264
+ short += self.newline
265
+ overflow = len(short) > BLOCK
266
+ if not overflow:
267
+ ids = prefix + short
268
+ slots = [len(prefix) + s for s in rel]
269
+ blocks = _blocks(0, len(prefix)) + [(len(prefix), len(ids))]
270
+ else:
271
+ catalogue = list(head)
272
+ for pos, o in enumerate(opts):
273
+ catalogue += self._option_label(pos, True) + list(o) + self.newline
274
+ doc_end = len(prefix) + len(catalogue)
275
+ rubric, rel = list(self.rubric_head), []
276
+ for pos in range(k):
277
+ rubric += self._option_label(pos, False) + self.arrow
278
+ rel.append(len(rubric) - 1)
279
+ rubric += self.newline
280
+ ids = prefix + catalogue + rubric
281
+ slots = [doc_end + s for s in rel]
282
+ blocks = _blocks(0, len(prefix)) + _blocks(len(prefix), doc_end) + _blocks(doc_end, len(ids))
283
+ if len(ids) > maximum:
284
+ raise InputBudgetError(f"the complete input needs {len(ids)} tokens (state {len(prefix)} + question/"
285
+ f"options/readout {len(ids) - len(prefix)}); the limit is {maximum} tokens (model "
286
+ f"maximum {CONTEXT_LIMIT}). Nothing was truncated: shorten the state, the "
287
+ f"question or the options.")
288
+ return Rendered(ids, len(prefix), slots, blocks, list(pieces["names"]), pieces["qtype"], overflow)
289
+
290
+ def render(self, state, question, max_len=None, prefix=None):
291
+ return self.assemble(self.pieces(state, question, prefix), max_len)
292
+
293
+
294
+ # -- calibration ----------------------------------------------------------------------------------
295
+ def check_temperature(t):
296
+ try:
297
+ t = float(t)
298
+ except (TypeError, ValueError):
299
+ raise ValueError(f"temperature must be a number, got {t!r}") from None
300
+ if not (math.isfinite(t) and t > 0):
301
+ raise ValueError(f"temperature must be finite and > 0, got {t!r}")
302
+ return t
303
+
304
+
305
+ def softmax_probabilities(scores, temperature):
306
+ """softmax(scores / T) in float64 (the reference computation)."""
307
+ z = np.asarray(scores, float) / float(temperature)
308
+ z = np.exp(z - z.max())
309
+ return z / z.sum()
310
+
311
+
312
+ def concentration(p):
313
+ k = len(p)
314
+ if k < 2:
315
+ return 1.0
316
+ ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
317
+ return float(np.clip(1.0 - ent / math.log(k), 0.0, 1.0))
318
+
319
+
320
+ def _sha256(path):
321
+ h = hashlib.sha256()
322
+ with open(path, "rb") as f:
323
+ for b in iter(lambda: f.read(1 << 22), b""):
324
+ h.update(b)
325
+ return h.hexdigest()
326
+
327
+
328
+ NOT_VERIFIED = ("README.md", "eval_results.json") # documentation / records: in manifest.json, not checked
329
+ NOT_VERIFIED_DIRS = ("assets/", "figures/", "validation/")
330
+
331
+
332
+ def verify_manifest(model_dir, only=None):
333
+ """Re-hash the files listed in manifest.json (all, or those whose path starts with one of ``only``).
334
+ Documentation and evaluation records (README.md, eval_results.json, assets/, figures/, validation/) are
335
+ recorded in the manifest but not checked here (not even when named in ``only``), so a card edit never
336
+ makes the runtime refuse to load and a download without them still verifies."""
337
+ model_dir = Path(model_dir)
338
+ man = json.loads((model_dir / "manifest.json").read_text())
339
+ bad, missing, checked = [], [], 0
340
+ for name, rec in man["files"].items():
341
+ if name in NOT_VERIFIED or name.startswith(NOT_VERIFIED_DIRS):
342
+ continue
343
+ if only and not any(name == o or name.startswith(o.rstrip("/") + "/") for o in only):
344
+ continue
345
+ p = model_dir / name
346
+ if not p.exists():
347
+ missing.append(name)
348
+ elif _sha256(p) != rec["sha256"]:
349
+ bad.append(name)
350
+ checked += 1
351
+ return {"ok": not bad and not missing, "checked": checked, "bad": bad, "missing": missing}
352
+
353
+
354
+ class DecisionBase:
355
+ """Backend-independent part. A backend implements ``_scores_many(rendered) -> list of list[float]``:
356
+ ``rendered`` is a non-empty list of Rendered that all share ONE state (identical prefix ids and
357
+ state blocks); it returns the raw float32 scores of every slot of every Rendered (slot order =
358
+ canonical option order), computing the state blocks once per call. Backends keep the most recent
359
+ state for the next call (exact prefix match only), so consecutive calls about the same state (e.g.
360
+ JSONL rows read from stdin) reuse it."""
361
+ backend = "base"
362
+
363
+ def _setup(self, model_dir, tokenizer_json, max_len=CONTEXT_LIMIT, temperature=None):
364
+ self.model_dir = Path(model_dir)
365
+ self.readout_config = json.loads((self.model_dir / "readout_config.json").read_text())
366
+ self.calibrated_temperature = check_temperature(self.readout_config["temperatures"]["global"])
367
+ self.default_temperature = self.calibrated_temperature if temperature is None else check_temperature(temperature)
368
+ self.encode = TextEncoder(tokenizer_json)
369
+ self.renderer = Renderer(self.encode, self.readout_config, max_len=max_len)
370
+
371
+ def _scores_many(self, rendered):
372
+ raise NotImplementedError
373
+
374
+ def close(self):
375
+ pass
376
+
377
+ def __enter__(self):
378
+ return self
379
+
380
+ def __exit__(self, *exc):
381
+ self.close()
382
+
383
+ def _score_all(self, rendered):
384
+ """Raw scores (lists of floats, canonical order) for Rendered sharing one state; one backend call."""
385
+ if not rendered:
386
+ return []
387
+ p = rendered[0].prefix_len
388
+ prefix, sblocks = rendered[0].ids[:p], rendered[0].state_blocks
389
+ for r in rendered[1:]:
390
+ if r.prefix_len != p or r.ids[:p] != prefix or r.state_blocks != sblocks:
391
+ raise ValueError("all questions of one backend call must share the same state")
392
+ scores = self._scores_many(rendered)
393
+ if len(scores) != len(rendered) or any(len(s) != len(r.slots) for s, r in zip(scores, rendered)):
394
+ raise RuntimeError("backend returned a wrong number of scores")
395
+ return [[float(x) for x in s] for s in scores]
396
+
397
+ def score_pieces(self, pieces_list, max_len=None):
398
+ """Raw scores for pre-tokenised questions sharing one state: dicts {"prefix", "head", "opts",
399
+ "names", "qtype"} of token ids (see Renderer.pieces). For parity checks; no temperature."""
400
+ return self._score_all([self.renderer.assemble(p, max_len) for p in pieces_list])
401
+
402
+ def _result(self, r, scores, temperature):
403
+ t = self.default_temperature if temperature is None else check_temperature(temperature)
404
+ bad = [n for n, x in zip(r.names, scores) if not math.isfinite(x)]
405
+ if bad:
406
+ raise NonFiniteScoreError(f"non-finite decision scores for option(s) {bad} ({len(r.ids)} input tokens); "
407
+ f"refusing to return probabilities. Check the weights file / engine build.")
408
+ p = softmax_probabilities(scores, t)
409
+ if not np.all(np.isfinite(p)):
410
+ raise NonFiniteScoreError("non-finite probabilities")
411
+ i = int(p.argmax())
412
+ return {"answer": r.names[i], "probabilities": dict(zip(r.names, p.tolist())),
413
+ "scores": dict(zip(r.names, scores)), "temperature": t, "top_probability": float(p[i]),
414
+ "entropy_concentration": concentration(p), "input_tokens": len(r.ids),
415
+ "state_tokens": r.prefix_len, "head_tokens": r.head_tokens, "blocks": len(r.blocks),
416
+ "catalogue_overflow": r.catalogue_overflow, "model": MODEL_NAME, "backend": self.backend}
417
+
418
+ def decide(self, state, question, options=None, qtype=None, category=None, temperature=None, head_max=None,
419
+ max_len=None):
420
+ """Score one question about ``state``.
421
+
422
+ Returns {"answer", "probabilities" {option: p}, "scores" {option: logit(yes)-logit(no)},
423
+ "temperature", "top_probability", "entropy_concentration", "input_tokens", "state_tokens",
424
+ "head_tokens", "blocks", "catalogue_overflow", "model", "backend"}. ``temperature`` overrides
425
+ the calibrated global temperature (1.0 = uncalibrated scores). ``category`` and ``head_max`` are
426
+ accepted for compatibility with the 0.8B v3 runtime and the jev-style package, and ignored: this
427
+ model has one global temperature and no separate question/options budget. Raises InputBudgetError
428
+ (never truncates), QuestionError or NonFiniteScoreError."""
429
+ q = make_question(question, options, qtype)
430
+ r = self.renderer.render(state, q, max_len)
431
+ return self._result(r, self._score_all([r])[0], temperature)
432
+
433
+ def score_many(self, state, questions, category=None, temperature=None, head_max=None, max_len=None):
434
+ """Several questions about ONE state: the state is tokenised and computed once and reused for
435
+ every question (one backend call). ``questions``: dicts {"t","ins","crit"} (or anything
436
+ make_question accepts). Results in order, identical to calling decide() per question. All
437
+ questions are rendered and budget-checked before any scoring. ``category`` / ``head_max``: see
438
+ decide() (accepted and ignored)."""
439
+ qs = [make_question(q) for q in questions]
440
+ if not qs:
441
+ return []
442
+ prefix = self.renderer.prefix_ids(state)
443
+ rs = [self.renderer.render(state, q, max_len, prefix=prefix) for q in qs]
444
+ return [self._result(r, sc, temperature) for r, sc in zip(rs, self._score_all(rs))]
445
+
446
+ decide_many = score_many
447
+
448
+
449
+ def base_arg_parser(description):
450
+ ap = argparse.ArgumentParser(description=description)
451
+ ap.add_argument("--model-dir", default=str(HERE), help="folder with the weights and readout_config.json")
452
+ ap.add_argument("--state", help="state as plain text")
453
+ ap.add_argument("--state-json", help="state as a JSON value")
454
+ ap.add_argument("--question", help="question text (or a JSON question {'t','ins','crit'})")
455
+ ap.add_argument("--options", help="JSON: {name: description} or [names] (choice); [levels] (score)")
456
+ ap.add_argument("--qtype", choices=QTYPES)
457
+ ap.add_argument("--temperature", type=float,
458
+ help="override the calibrated global temperature of readout_config.json (1.0 = raw scores)")
459
+ ap.add_argument("--max-len", type=int, default=CONTEXT_LIMIT,
460
+ help=f"total token budget (state + question + options + readout), at most {CONTEXT_LIMIT}")
461
+ ap.add_argument("--jsonl", help="batch mode: input JSON lines {id?, state, question, options?, qtype?, "
462
+ "temperature?} ('-' = stdin); one JSON result per line on stdout. Consecutive "
463
+ "rows with an identical state share one state computation")
464
+ ap.add_argument("--verify", action="store_true", help="check sha256 of the files in manifest.json first")
465
+ return ap
466
+
467
+
468
+ def _jsonl_groups(src, streaming):
469
+ """(line number, record or error text) grouped into runs of consecutive rows with one state. From a
470
+ file up to JSONL_GROUP_MAX rows are grouped; from stdin every row is its own group (answered at once;
471
+ the backend's kept state still makes consecutive identical states cheap)."""
472
+ group, key = [], None
473
+ for n, line in enumerate(src):
474
+ if not line.strip():
475
+ continue
476
+ try:
477
+ rec = json.loads(line)
478
+ if not isinstance(rec, dict):
479
+ raise ValueError("a JSONL row must be a JSON object")
480
+ k = serialize_state(rec.get("state", ""))
481
+ except ValueError as e:
482
+ if group:
483
+ yield group
484
+ group, key = [], None
485
+ yield [(n, f"{type(e).__name__}: {e}")]
486
+ continue
487
+ if group and (k != key or len(group) >= JSONL_GROUP_MAX):
488
+ yield group
489
+ group = []
490
+ group.append((n, rec))
491
+ key = k
492
+ if streaming:
493
+ yield group
494
+ group, key = [], None
495
+ if group:
496
+ yield group
497
+
498
+
499
+ def _run_group(engine, group, args):
500
+ """Score one group of JSONL rows sharing a state. Returns (output rows, non-finite count)."""
501
+ out, todo = {}, []
502
+ prefix = None
503
+ for n, rec in group:
504
+ if isinstance(rec, str):
505
+ out[n] = {"id": n, "error": rec}
506
+ continue
507
+ rid = rec.get("id", n)
508
+ try:
509
+ if "question" not in rec:
510
+ raise QuestionError("row has no 'question'")
511
+ q = make_question(rec["question"], options=rec.get("options"), qtype=rec.get("qtype"))
512
+ t = rec.get("temperature", args.temperature)
513
+ t = None if t is None else check_temperature(t)
514
+ if prefix is None:
515
+ prefix = engine.renderer.prefix_ids(rec.get("state", ""))
516
+ todo.append((n, rid, engine.renderer.render(rec.get("state", ""), q, prefix=prefix), t))
517
+ except (InputBudgetError, QuestionError, ValueError, TypeError, AttributeError) as e:
518
+ out[n] = {"id": rid, "error": f"{type(e).__name__}: {e}"}
519
+ nonfinite = 0
520
+ if todo:
521
+ scores = engine._score_all([r for _, _, r, _ in todo])
522
+ for (n, rid, r, t), sc in zip(todo, scores):
523
+ try:
524
+ out[n] = {"id": rid, **engine._result(r, sc, t)}
525
+ except NonFiniteScoreError as e:
526
+ nonfinite += 1
527
+ print(f"ERROR row {rid}: NonFiniteScoreError: {e}", file=sys.stderr, flush=True)
528
+ out[n] = {"id": rid, "error": f"NonFiniteScoreError: {e}"}
529
+ return [out[n] for n, _ in group], nonfinite
530
+
531
+
532
+ def run_cli(args, engine):
533
+ """Exit status: 0 = ok (JSONL rows with input errors carry an "error" field), 2 = input error
534
+ (single question), 3 = at least one non-finite score (refused, see stderr)."""
535
+ if args.jsonl:
536
+ streaming = args.jsonl == "-"
537
+ src = sys.stdin if streaming else open(args.jsonl, encoding="utf-8")
538
+ nonfinite = 0
539
+ try:
540
+ for group in _jsonl_groups(src, streaming):
541
+ rows, bad = _run_group(engine, group, args)
542
+ nonfinite += bad
543
+ for row in rows:
544
+ print(json.dumps(row, ensure_ascii=False), flush=True)
545
+ finally:
546
+ if not streaming:
547
+ src.close()
548
+ return 3 if nonfinite else 0
549
+ if args.question is None:
550
+ raise SystemExit("--question (or --jsonl) is required")
551
+ try:
552
+ state = json.loads(args.state_json) if args.state_json is not None else (args.state or "")
553
+ question = args.question
554
+ if question.lstrip().startswith("{"):
555
+ try: # a JSON question {'t','ins','crit'}; else plain text
556
+ parsed = json.loads(question)
557
+ except ValueError:
558
+ parsed = None
559
+ if isinstance(parsed, dict):
560
+ question = parsed
561
+ options = json.loads(args.options) if args.options else None
562
+ res = engine.decide(state, question, options=options, qtype=args.qtype, temperature=args.temperature)
563
+ except (InputBudgetError, QuestionError, ValueError, TypeError) as e: # NonFiniteScoreError is not a ValueError
564
+ print(f"error: {type(e).__name__}: {e}", file=sys.stderr)
565
+ return 2
566
+ except NonFiniteScoreError as e:
567
+ print(f"ERROR: NonFiniteScoreError: {e}", file=sys.stderr)
568
+ return 3
569
+ print(json.dumps(res, ensure_ascii=False, indent=2))
570
+ return 0
571
+ # ---------------------------------------------------------------------------- end of shared core
572
+
573
+
574
+ # ---------------------------------------------------------------------------------- MLX backend
575
+ # Repository layout (one runtime, two precisions):
576
+ #
577
+ # jev_style_decision_mlx.py, readout_config.json, release_config.json, manifest.json, ... shared
578
+ # bf16/ config.json, model.safetensors (bfloat16), macjev_norms_fp32.safetensors, tokenizer ...
579
+ # 8bit/ config.json, model.safetensors (affine 8-bit, group 64), macjev_norms_fp32.safetensors, tokenizer ...
580
+ #
581
+ # Either folder may be downloaded alone (plus the shared top-level files).
582
+ #
583
+ # Block-causal attention on mlx-lm: the input is fed to the model ONE protocol block per call on an mlx-lm
584
+ # prompt cache. The 6 full-attention layers use BlockKVCache, whose make_mask returns None, so the queries of
585
+ # the current block attend to every cached key plus every key of the block itself, with no causal mask inside
586
+ # the block. The 18 Gated-DeltaNet layers keep their recurrent cache (conv + recurrent state), causal by
587
+ # construction. The state blocks are computed once; each question then runs its block(s) on an independent
588
+ # copy of that cache, so questions never see each other and the state is reused (score_many, JSONL rows).
589
+ #
590
+ # Two corrections to stock mlx-lm 0.31.3 (both needed to match the HF / training numerics):
591
+ # 1. Gated-DeltaNet q/k normalisation: mlx-lm uses rms_norm(x, eps=1e-6) (= l2norm with eps Dk*1e-6); HF / fla /
592
+ # llama.cpp use l2norm(x, eps=1e-6). A copy of mlx-lm's GatedDeltaNet.__call__ with only that eps changed is
593
+ # swapped in per instance (never globally).
594
+ # 2. Qwen3.5 RMSNorm is x_hat * (1 + w). mlx-lm's converter adds the 1 in bf16, so the export stores bf16(1 + w).
595
+ # The exact FP32 (1 + w) ships as the sidecar macjev_norms_fp32.safetensors (ignored by stock mlx-lm); the
596
+ # runtime installs it and computes those norms in FP32 (cast back to the input dtype, as HF does).
597
+ # Both depend on mlx-lm internals, so the runtime checks the sha256 of the mlx-lm source it patches or relies on
598
+ # and refuses to run on any other mlx-lm version (install mlx-lm==0.31.3). The sidecar and readout_config.json are
599
+ # mandatory: the runtime never falls back to stock norms or to T = 1.
600
+ REPO_ID = "chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX"
601
+ PRECISIONS = ("bf16", "8bit")
602
+ DEFAULT_PRECISION = "bf16"
603
+ # activation dtype per precision: None = the checkpoint's native activations (bf16 residual stream), the
604
+ # setting the release scoring used for both precisions (release_config.json -> runtime.precisions overrides)
605
+ VALIDATED_COMPUTE_DTYPE = {"bf16": None, "8bit": None}
606
+ MLX_CACHE_LIMIT_GIB = 2.0
607
+ MLX_LM_VERSION = "0.31.3"
608
+ NORM_SIDECAR = "macjev_norms_fp32.safetensors"
609
+ NORM_SIDECAR_FORMAT = "macjev-norms-fp32-v1"
610
+ SHIFTED_NORMS = ("input_layernorm", "post_attention_layernorm", "q_norm", "k_norm")
611
+ # sha256 of inspect.getsource(...) in mlx-lm 0.31.3 for every function the MLX path patches (GatedDeltaNet) or relies
612
+ # on for the block mask (a None mask from BlockKVCache must reach scaled_dot_product_attention unchanged)
613
+ MLX_LM_SOURCE_SHA256 = {
614
+ "qwen3_5.GatedDeltaNet.__call__": "9f27bdc613912fa9b3b0509d8e3f684ac672810bf45e29831f5815a1c3b43091",
615
+ "qwen3_5.Qwen3_5TextModel.__call__": "45fd379f8885e93b4ba685c3d6c2aa1b891d535bb00fd042c6b1e56613345b6d",
616
+ "base.create_attention_mask": "a8ebacf963b3f96e9ad838a7a870d1ef6da33aec5233b133b0d0c744c3b33e5e",
617
+ "qwen3_5.Attention.__call__": "a938424855f1b919d1bcfd62b105e10723f7de295f5ee51a19924b502ddc86a6",
618
+ "base.scaled_dot_product_attention": "8deba4a8a8fb4e29f1e69ea81701da68f08f90656e1421a1e7a82332aadc41c1",
619
+ }
620
+
621
+
622
+ class MLXVersionError(RuntimeError):
623
+ """The installed mlx-lm is not the version whose internals the runtime was validated against."""
624
+
625
+
626
+ def check_mlx_lm():
627
+ """Raise MLXVersionError unless the mlx-lm source this runtime patches is byte-identical to mlx-lm 0.31.3."""
628
+ import inspect
629
+ import mlx_lm
630
+ from mlx_lm.models import base, qwen3_5
631
+ objs = {"qwen3_5.GatedDeltaNet.__call__": qwen3_5.GatedDeltaNet.__call__,
632
+ "qwen3_5.Qwen3_5TextModel.__call__": qwen3_5.Qwen3_5TextModel.__call__,
633
+ "base.create_attention_mask": base.create_attention_mask,
634
+ "qwen3_5.Attention.__call__": qwen3_5.Attention.__call__,
635
+ "base.scaled_dot_product_attention": base.scaled_dot_product_attention}
636
+ changed = [k for k, f in objs.items()
637
+ if hashlib.sha256(inspect.getsource(f).encode()).hexdigest() != MLX_LM_SOURCE_SHA256[k]]
638
+ if changed:
639
+ raise MLXVersionError(
640
+ f"this runtime needs mlx-lm=={MLX_LM_VERSION} (installed: {getattr(mlx_lm, '__version__', '?')}): it "
641
+ f"corrects the Qwen3.5 Gated-DeltaNet q/k norm and the (1 + w) RMSNorm weights inside mlx-lm, and the "
642
+ f"source it patches differs ({', '.join(changed)}). Install the tested version with: "
643
+ f"pip install 'mlx-lm=={MLX_LM_VERSION}'")
644
+
645
+
646
+ # Third-party code: _gdn_hf_call below is a modified copy of GatedDeltaNet.__call__ from mlx_lm/models/qwen3_5.py,
647
+ # mlx-lm 0.31.3 (https://github.com/ml-explore/mlx-lm), MIT License (mlx-lm LICENSE: "Copyright © 2023 Apple Inc.";
648
+ # qwen3_5.py header: "Copyright © 2026 Apple Inc."). Changes:
649
+ # q/k rms_norm eps 1e-6 -> 1e-6 / Dk (HF-exact l2norm), sharding branches removed (refused), module-level function.
650
+ # The MIT copyright and permission notice is reproduced in THIRD_PARTY_NOTICES.md of this repository.
651
+ def _gdn_hf_call(self, inputs, mask=None, cache=None):
652
+ """mlx-lm 0.31.3 qwen3_5.GatedDeltaNet.__call__ with HF-exact q/k l2norm (eps 1e-6). No sharding."""
653
+ import mlx.core as mx
654
+ import mlx.nn as nn
655
+ from mlx_lm.models.gated_delta import gated_delta_update
656
+ if self.sharding_group is not None:
657
+ raise ValueError("sharded Gated-DeltaNet is not supported by this runtime")
658
+ B, S, _ = inputs.shape
659
+ qkv = self.in_proj_qkv(inputs)
660
+ z = self.in_proj_z(inputs).reshape(B, S, self.num_v_heads, self.head_v_dim)
661
+ b = self.in_proj_b(inputs)
662
+ a = self.in_proj_a(inputs)
663
+ if cache is not None and cache[0] is not None:
664
+ conv_state = cache[0]
665
+ else:
666
+ conv_state = mx.zeros((B, self.conv_kernel_size - 1, self.conv_dim), dtype=inputs.dtype)
667
+ if mask is not None:
668
+ qkv = mx.where(mask[..., None], qkv, 0)
669
+ conv_input = mx.concatenate([conv_state, qkv], axis=1)
670
+ if cache is not None:
671
+ n_keep = self.conv_kernel_size - 1
672
+ if cache.lengths is not None:
673
+ ends = mx.clip(cache.lengths, 0, S)
674
+ positions = (ends[:, None] + mx.arange(n_keep))[..., None]
675
+ cache[0] = mx.take_along_axis(conv_input, positions, axis=1)
676
+ else:
677
+ cache[0] = mx.contiguous(conv_input[:, -n_keep:, :])
678
+ conv_out = nn.silu(self.conv1d(conv_input))
679
+ q, k, v = [t.reshape(B, S, h, d) for t, h, d in zip(
680
+ mx.split(conv_out, [self.key_dim, 2 * self.key_dim], -1),
681
+ [self.num_k_heads, self.num_k_heads, self.num_v_heads],
682
+ [self.head_k_dim, self.head_k_dim, self.head_v_dim])]
683
+ state = cache[1] if cache else None
684
+ dk = k.shape[-1]
685
+ inv_scale = dk ** -0.5
686
+ # l2norm(x, 1e-6) == rms_norm(x, eps=1e-6/Dk) / sqrt(Dk); q additionally carries the 1/sqrt(Dk) scale
687
+ q = (inv_scale ** 2) * mx.fast.rms_norm(q, None, 1e-6 / dk)
688
+ k = inv_scale * mx.fast.rms_norm(k, None, 1e-6 / dk)
689
+ out, state = gated_delta_update(q, k, v, a, b, self.A_log, self.dt_bias, state, mask,
690
+ use_kernel=not self.training)
691
+ if cache is not None:
692
+ cache[1] = state
693
+ cache.advance(S)
694
+ out = self.norm(out, z)
695
+ return self.out_proj(out.reshape(B, S, -1))
696
+
697
+
698
+ def patch_gdn_l2norm(inner):
699
+ """Swap every GatedDeltaNet instance of ``inner`` to the HF-exact l2norm class (per instance). -> count."""
700
+ from mlx_lm.models.qwen3_5 import GatedDeltaNet
701
+ cls = type("GatedDeltaNetHFL2", (GatedDeltaNet,), {"__call__": _gdn_hf_call})
702
+ n = 0
703
+ for layer in inner.layers:
704
+ if getattr(layer, "is_linear", False):
705
+ if type(layer.linear_attn) is not GatedDeltaNet:
706
+ raise TypeError(f"unexpected Gated-DeltaNet class {type(layer.linear_attn).__name__}")
707
+ layer.linear_attn.__class__ = cls
708
+ n += 1
709
+ return n
710
+
711
+
712
+ def load_norm_sidecar(path):
713
+ """{mlx parameter name: FP32 (1 + w)} from macjev_norms_fp32.safetensors (format checked)."""
714
+ import mlx.core as mx
715
+ arrays, meta = mx.load(str(path), return_metadata=True)
716
+ if (meta or {}).get("format") != NORM_SIDECAR_FORMAT:
717
+ raise ValueError(f"{path} is not a {NORM_SIDECAR_FORMAT} norm sidecar (metadata {meta!r})")
718
+ mx.eval(arrays) # read now (mx.load is lazy)
719
+ return arrays
720
+
721
+
722
+ def apply_exact_norms(inner, norms):
723
+ """Install FP32 (1 + w) into every shifted RMSNorm of ``inner`` (per-instance class swap). -> count.
724
+ Refuses a sidecar that misses a norm, names an unknown module or does not belong to these weights."""
725
+ import mlx.core as mx
726
+ import mlx.nn as nn
727
+
728
+ def _call(self, x):
729
+ return mx.fast.rms_norm(x.astype(mx.float32), self.weight, self.eps).astype(x.dtype)
730
+
731
+ cls = type("RMSNormFP32Weight", (nn.RMSNorm,), {"__call__": _call})
732
+ modules = dict(inner.named_modules())
733
+ seen = set()
734
+ for name, w in norms.items():
735
+ mod_name = name[:-len(".weight")] if name.endswith(".weight") else name
736
+ mod = modules.get(mod_name)
737
+ if mod is None or type(mod) is not nn.RMSNorm:
738
+ raise KeyError(f"norm sidecar entry {name} is not an RMSNorm of this model")
739
+ w = w.astype(mx.float32)
740
+ if w.shape != mod.weight.shape:
741
+ raise ValueError(f"norm sidecar shape mismatch at {name}")
742
+ # the sidecar must be the (1 + w) of THESE weights (up to the bf16 rounding it corrects)
743
+ if float(mx.abs(w - mod.weight.astype(mx.float32)).max()) > 2 ** -7 * float(mx.abs(w).max()) + 1e-6:
744
+ raise ValueError(f"norm sidecar does not match the model weights at {name}")
745
+ mod.weight = w
746
+ mod.__class__ = cls
747
+ seen.add(mod_name)
748
+ missing = [k for k, m in modules.items() if type(m) is nn.RMSNorm and k not in seen
749
+ and (k == "norm" or k.endswith(SHIFTED_NORMS))]
750
+ if missing:
751
+ raise KeyError(f"norm sidecar misses {missing[:3]}")
752
+ mx.eval(inner.parameters())
753
+ return len(seen)
754
+
755
+
756
+ def _block_kv_class():
757
+ from mlx_lm.models.cache import KVCache
758
+
759
+ class BlockKVCache(KVCache):
760
+ """KV cache whose make_mask is None: the current block attends to all cached keys and to itself
761
+ (Qwen3_5TextModel builds one mask from the first full-attention layer's cache)."""
762
+
763
+ def make_mask(self, *args, **kwargs):
764
+ return None
765
+
766
+ return BlockKVCache
767
+
768
+
769
+ def copy_cache(cache):
770
+ """Independent copy of a prompt cache (KV + ArraysCache). KV copies are trimmed to ``offset`` (the next
771
+ update re-allocates by concatenation); ArraysCache gets a new list (GDN layers assign new arrays)."""
772
+ import copy
773
+ from mlx_lm.models.cache import ArraysCache, KVCache
774
+ out = []
775
+ for c in cache:
776
+ n = copy.copy(c)
777
+ if isinstance(c, KVCache):
778
+ if c.keys is not None:
779
+ n.keys = c.keys[..., :c.offset, :]
780
+ n.values = c.values[..., :c.offset, :]
781
+ elif isinstance(c, ArraysCache):
782
+ n.cache = list(c.cache)
783
+ if c.lengths is not None or c.left_padding is not None:
784
+ raise ValueError("batched ArraysCache state is not supported")
785
+ else:
786
+ raise TypeError(f"unsupported cache type {type(c).__name__}")
787
+ out.append(n)
788
+ return out
789
+
790
+
791
+ def _set_mlx_cache_limit(mx, gib):
792
+ """Bound MLX's freed-buffer cache (process-wide); never raises a lower limit set by the host."""
793
+ if gib is None:
794
+ return None
795
+ want = int(float(gib) * (1 << 30))
796
+ prev = mx.set_cache_limit(want)
797
+ if prev is not None and prev < want:
798
+ mx.set_cache_limit(prev)
799
+ return int(prev)
800
+ return want
801
+
802
+
803
+ def _weights_precision(weights_dir):
804
+ """'bf16' or '8bit' (or '<n>bit') from the MLX config.json of a weights folder."""
805
+ cfg = json.loads((Path(weights_dir) / "config.json").read_text())
806
+ q = cfg.get("quantization") or cfg.get("quantization_config")
807
+ return f"{int(q['bits'])}bit" if q else "bf16"
808
+
809
+
810
+ def _require_weights(w, precision, repo):
811
+ need = ["config.json", "model.safetensors", "tokenizer.json", NORM_SIDECAR]
812
+ miss = [n for n in need if not (w / n).is_file()]
813
+ if miss:
814
+ extra = (f" {NORM_SIDECAR} holds the exact FP32 (1 + w) RMSNorm weights; without it mlx-lm would use "
815
+ f"bf16-rounded norms, so the runtime refuses to load." if NORM_SIDECAR in miss else "")
816
+ raise FileNotFoundError(f"{w} is missing {', '.join(miss)}.{extra} Download the folder with: "
817
+ f"hf download {REPO_ID} --include \"{precision}/*\" --local-dir {repo}")
818
+
819
+
820
+ def resolve_model_dirs(model_dir=HERE, precision=None):
821
+ """-> (repo_dir, weights_dir, precision).
822
+
823
+ ``model_dir`` is the repository folder (holding bf16/ and/or 8bit/ next to readout_config.json),
824
+ or one precision folder itself (then ``precision`` defaults to the precision stored there).
825
+ ``precision`` None means "bf16" for a repository folder."""
826
+ if precision is not None and precision not in PRECISIONS:
827
+ raise ValueError(f"precision must be one of {PRECISIONS}, got {precision!r}")
828
+ d = Path(model_dir).resolve()
829
+ # a precision folder given directly (the repository root also holds a config.json, a copy of bf16/config.json,
830
+ # so a folder counts as a precision folder only if it has its own weights or no bf16/ / 8bit/ subfolder)
831
+ if (d / "config.json").is_file() and ((d / "model.safetensors").is_file()
832
+ or not any((d / p).is_dir() for p in PRECISIONS)):
833
+ found = _weights_precision(d)
834
+ if precision is not None and precision != found:
835
+ raise ValueError(f"{d} holds {found} weights, not {precision}")
836
+ repo = d if (d / "readout_config.json").is_file() else d.parent
837
+ _require_weights(d, found, repo)
838
+ return repo, d, found
839
+ precision = precision or DEFAULT_PRECISION
840
+ w = d / precision
841
+ if not ((w / "config.json").is_file() and (w / "model.safetensors").is_file()):
842
+ have = [p for p in PRECISIONS if (d / p / "config.json").is_file() and (d / p / "model.safetensors").is_file()]
843
+ msg = f"the {precision} weights are not in {d} (expected {precision}/config.json and {precision}/model.safetensors)."
844
+ if have:
845
+ msg += f" This folder has only: {', '.join(p + '/' for p in have)}; use --precision {have[0]} (precision=\"{have[0]}\"), or"
846
+ msg += f" download them with: hf download {REPO_ID} --include \"{precision}/*\" --local-dir {d}"
847
+ raise FileNotFoundError(msg)
848
+ found = _weights_precision(w)
849
+ if found != precision:
850
+ raise ValueError(f"{w} holds {found} weights, not {precision}")
851
+ _require_weights(w, precision, d)
852
+ return d, w, precision
853
+
854
+
855
+ def verify_repo(repo_dir, weights_dir):
856
+ """manifest check of the shared top-level files and of the selected precision folder only (the other
857
+ precision may be absent). Documentation and records (README.md, figures/, validation/) are skipped as in
858
+ verify_manifest."""
859
+ repo_dir, weights_dir = Path(repo_dir), Path(weights_dir)
860
+ if weights_dir == repo_dir:
861
+ return verify_manifest(repo_dir)
862
+ sub = weights_dir.relative_to(repo_dir).as_posix()
863
+ names = json.loads((repo_dir / "manifest.json").read_text())["files"]
864
+ if not any(n.startswith(sub + "/") for n in names):
865
+ return {"ok": False, "checked": 0, "bad": [], "missing": [f"{sub}/ (not listed in manifest.json)"]}
866
+ res = verify_manifest(repo_dir, only=[n for n in names if "/" not in n] + [sub + "/"])
867
+ res["precision_folder"] = sub + "/"
868
+ return res
869
+
870
+
871
+ def release_compute_dtype(repo_dir, precision):
872
+ """Activation dtype of ``precision`` (release_config.json -> runtime.precisions, else the built-in default)."""
873
+ p = Path(repo_dir) / "release_config.json"
874
+ if p.is_file():
875
+ per = (json.loads(p.read_text()).get("runtime", {}).get("precisions") or {}).get(precision)
876
+ if per is not None and "mlx_compute_dtype" in per:
877
+ return per["mlx_compute_dtype"]
878
+ return VALIDATED_COMPUTE_DTYPE[precision]
879
+
880
+
881
+ class _MLXEngine:
882
+ """Block-causal v2 scoring on mlx-lm (see the MLX backend comment above). Stateless apart from the last
883
+ state cache (kept so consecutive requests with the same state reuse it)."""
884
+
885
+ def __init__(self, weights_dir, yes_id, no_id, compute_dtype=None, cache_limit_gib=MLX_CACHE_LIMIT_GIB):
886
+ import mlx.core as mx
887
+ check_mlx_lm()
888
+ from mlx_lm.utils import load_model
889
+ self.mx = mx
890
+ self.cache_limit_bytes = _set_mlx_cache_limit(mx, cache_limit_gib)
891
+ norms = load_norm_sidecar(Path(weights_dir) / NORM_SIDECAR)
892
+ self.model, self.config = load_model(Path(weights_dir))
893
+ if compute_dtype:
894
+ self.model.set_dtype(getattr(mx, compute_dtype))
895
+ self.compute_dtype = compute_dtype
896
+ lm = getattr(self.model, "language_model", self.model)
897
+ self.inner = lm.model
898
+ args = getattr(lm, "args", None) or getattr(self.model, "args", None)
899
+ if not bool(getattr(args, "tie_word_embeddings", True)) or hasattr(lm, "lm_head"):
900
+ raise ValueError("expected tied input/output embeddings")
901
+ self.n_gdn_patched = patch_gdn_l2norm(self.inner)
902
+ self.n_norms_exact = apply_exact_norms(self.inner, norms)
903
+ self.n_full_attention = sum(1 for layer in self.inner.layers if not layer.is_linear)
904
+ w = self.inner.embed_tokens(mx.array([int(yes_id), int(no_id)])) # dequantised rows if quantised
905
+ w = w.astype(mx.float32)
906
+ self.direction = w[0] - w[1]
907
+ mx.eval(self.direction)
908
+ self._state = None # (tuple(prefix ids), cache) of the last state
909
+
910
+ def _make_cache(self):
911
+ from mlx_lm.models.cache import ArraysCache
912
+ kv = _block_kv_class()
913
+ return [ArraysCache(size=2) if layer.is_linear else kv() for layer in self.inner.layers]
914
+
915
+ def _eval_cache(self, cache):
916
+ arrays = []
917
+ for c in cache:
918
+ if hasattr(c, "keys"):
919
+ if c.keys is not None:
920
+ arrays += [c.keys, c.values]
921
+ else:
922
+ arrays += [a for a in c.cache if a is not None]
923
+ self.mx.eval(arrays)
924
+
925
+ def state_cache(self, prefix, blocks):
926
+ """Cache after the state blocks (computed once; reused while the state is unchanged). -> (cache, reused)"""
927
+ key = tuple(prefix)
928
+ if self._state is not None and self._state[0] == key:
929
+ return self._state[1], True
930
+ self._state = None
931
+ cache = self._make_cache()
932
+ for a, b in blocks:
933
+ self.inner(self.mx.array(prefix[a:b])[None], cache=cache)
934
+ self._eval_cache(cache)
935
+ self._state = (key, cache)
936
+ return cache, False
937
+
938
+ def question_scores(self, ids, slots, qblocks, state_cache):
939
+ """Question block(s) on a copy of the state cache -> raw FP32 scores, one per slot (rendered order)."""
940
+ mx = self.mx
941
+ cache = copy_cache(state_cache)
942
+ found = {}
943
+ for i, (a, b) in enumerate(qblocks):
944
+ h = self.inner(mx.array(ids[a:b])[None], cache=cache)[0]
945
+ inside = [(j, s) for j, s in enumerate(slots) if a <= s < b]
946
+ if inside:
947
+ hs = h[mx.array([s - a for _, s in inside])].astype(mx.float32)
948
+ v = hs @ self.direction
949
+ mx.eval(v)
950
+ for (j, _), x in zip(inside, v.tolist()):
951
+ found[j] = x
952
+ if i + 1 < len(qblocks):
953
+ self._eval_cache(cache)
954
+ if len(found) != len(slots):
955
+ raise RuntimeError("a verdict slot lies outside the question blocks")
956
+ return [float(found[j]) for j in range(len(slots))]
957
+
958
+ def reset(self):
959
+ self._state = None
960
+
961
+ def close(self):
962
+ self._state = None
963
+ self.mx.clear_cache()
964
+
965
+
966
+ def _check_blocks(r):
967
+ """The protocol invariants a backend relies on: blocks tile [0, n) in order, each <= BLOCK tokens, the
968
+ state ends on a block boundary, every slot lies in a question block."""
969
+ n, p, end = len(r.ids), r.prefix_len, 0
970
+ for a, b in r.blocks:
971
+ if a != end or not a < b <= n or b - a > BLOCK:
972
+ raise ValueError(f"invalid block layout {r.blocks}")
973
+ end = b
974
+ sb, qb = r.state_blocks, r.question_blocks
975
+ if end != n or len(sb) + len(qb) != len(r.blocks) or (sb and sb[-1][1] != p) or not qb:
976
+ raise ValueError(f"invalid block layout {r.blocks} (prefix {p}, {n} tokens)")
977
+ if any(not qb[0][0] <= s < n for s in r.slots):
978
+ raise ValueError("verdict slots must lie in the question blocks")
979
+
980
+
981
+ class JevStyleDecisionMLX(DecisionBase):
982
+ """Apple-silicon runtime on mlx / mlx-lm 0.31.3 (qwen3_5 model code, two numerics corrections).
983
+
984
+ ``model_dir``: the repository folder (default: the folder of this script) holding bf16/ and/or 8bit/ next to
985
+ readout_config.json, or one precision folder. ``precision``: "bf16" (default) or "8bit".
986
+ ``temperature``: overrides the calibrated global temperature for every call (default: readout_config.json).
987
+ ``compute_dtype``: "release" (default) = the activation dtype of the chosen precision in release_config.json
988
+ (native bf16 activations for both precisions); "float32" or None (native) override it.
989
+ ``max_len``: total token budget (<= 25,600). ``verify``: re-hash the files in manifest.json before loading.
990
+
991
+ The most recent state stays computed: score_many() and consecutive calls / JSONL rows with an identical
992
+ state reuse it (``last_timing["state_reused"]``)."""
993
+ backend = "mlx"
994
+
995
+ def __init__(self, model_dir=HERE, precision=None, temperature=None, max_len=CONTEXT_LIMIT,
996
+ compute_dtype="release", cache_limit_gib=MLX_CACHE_LIMIT_GIB, verify=False):
997
+ repo_dir, weights_dir, precision = resolve_model_dirs(model_dir, precision)
998
+ self.repo_dir, self.weights_dir, self.precision = repo_dir, weights_dir, precision
999
+ if not (repo_dir / "readout_config.json").is_file():
1000
+ raise FileNotFoundError(f"readout_config.json (readout + calibration temperature) is not in {repo_dir}; "
1001
+ f"the runtime does not guess a temperature. Download it with: hf download "
1002
+ f"{REPO_ID} readout_config.json release_config.json --local-dir {repo_dir}")
1003
+ if verify:
1004
+ res = verify_repo(repo_dir, weights_dir)
1005
+ if not res["ok"]:
1006
+ raise RuntimeError(f"integrity check failed: {res}")
1007
+ self._setup(repo_dir, weights_dir / "tokenizer.json", max_len, temperature)
1008
+ if compute_dtype == "release":
1009
+ compute_dtype = release_compute_dtype(repo_dir, precision)
1010
+ if compute_dtype not in (None, "float32", "bfloat16", "float16"):
1011
+ raise ValueError(f"unsupported compute_dtype {compute_dtype!r}")
1012
+ self.engine = _MLXEngine(weights_dir, self.renderer.yes, self.renderer.no, compute_dtype, cache_limit_gib)
1013
+ self.compute_dtype = compute_dtype
1014
+ self.last_timing = {}
1015
+
1016
+ def _scores_many(self, rendered):
1017
+ import time
1018
+ for r in rendered:
1019
+ _check_blocks(r)
1020
+ r0 = rendered[0]
1021
+ t0 = time.perf_counter()
1022
+ cache, reused = self.engine.state_cache(r0.ids[:r0.prefix_len], r0.state_blocks)
1023
+ t1 = time.perf_counter()
1024
+ out = [self.engine.question_scores(r.ids, r.slots, r.question_blocks, cache) for r in rendered]
1025
+ self.last_timing = {"state_tokens": r0.prefix_len, "state_reused": reused, "state_ms": (t1 - t0) * 1000,
1026
+ "questions": len(rendered), "questions_ms": (time.perf_counter() - t1) * 1000}
1027
+ return out
1028
+
1029
+ def close(self):
1030
+ self.engine.close()
1031
+
1032
+
1033
+ def main(argv=None):
1034
+ ap = base_arg_parser(f"{MODEL_NAME}: typed decisions with MLX (Apple silicon)")
1035
+ for a in ap._actions:
1036
+ if a.dest == "model_dir":
1037
+ a.help = ("repository folder holding bf16/ and/or 8bit/ next to readout_config.json (default: the "
1038
+ "folder of this script), or one precision folder")
1039
+ ap.add_argument("--precision", choices=PRECISIONS, default=None,
1040
+ help="weights to load: bf16 (default) or 8bit (affine 8-bit, group 64)")
1041
+ ap.add_argument("--compute-dtype", default="release", choices=["release", "float32", "native"],
1042
+ help="activation dtype: the release setting of the precision (release = native), float32, or "
1043
+ "the checkpoint's native dtype")
1044
+ ap.add_argument("--cache-limit-gib", type=float, default=MLX_CACHE_LIMIT_GIB)
1045
+ args = ap.parse_args(argv)
1046
+ cdt = None if args.compute_dtype == "native" else args.compute_dtype
1047
+ try:
1048
+ engine = JevStyleDecisionMLX(args.model_dir, precision=args.precision, max_len=args.max_len,
1049
+ compute_dtype=cdt, cache_limit_gib=args.cache_limit_gib, verify=args.verify)
1050
+ except (FileNotFoundError, ValueError, MLXVersionError, RuntimeError, ImportError) as e:
1051
+ raise SystemExit(f"error: {e}")
1052
+ try:
1053
+ return run_cli(args, engine)
1054
+ finally:
1055
+ engine.close()
1056
+
1057
+
1058
+ if __name__ == "__main__":
1059
+ raise SystemExit(main())
manifest.json ADDED
@@ -0,0 +1,190 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-manifest-v1",
3
+ "repo": "chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX",
4
+ "created_unix": 1790476271.175225,
5
+ "files": {
6
+ "8bit/chat_template.jinja": {
7
+ "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80",
8
+ "bytes": 7755
9
+ },
10
+ "8bit/config.json": {
11
+ "sha256": "8865dd86a4e561e6b036667cd0d5c1c39ad12527b3feda7c585cf116d890477a",
12
+ "bytes": 2205
13
+ },
14
+ "8bit/generation_config.json": {
15
+ "sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73",
16
+ "bytes": 116
17
+ },
18
+ "8bit/macjev_norms_fp32.safetensors": {
19
+ "sha256": "02f08c422228929c44dd84856678ea95d41401138aa56b1b771a654376d3e51c",
20
+ "bytes": 420112
21
+ },
22
+ "8bit/model.safetensors": {
23
+ "sha256": "184b4dda2f1285fe95a00a6271b364c861d370fbf913030f0259b2423f88a740",
24
+ "bytes": 2000043057
25
+ },
26
+ "8bit/model.safetensors.index.json": {
27
+ "sha256": "54d94d008c07e39846c0b454f68055622bb635a0623b51bce86d724e18733aab",
28
+ "bytes": 60786
29
+ },
30
+ "8bit/tokenizer.json": {
31
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
32
+ "bytes": 19989325
33
+ },
34
+ "8bit/tokenizer_config.json": {
35
+ "sha256": "95c557768e6b88a7128befc7bfd3c7de50e5d51af9b8b33a9f4dee0e04f99679",
36
+ "bytes": 1161
37
+ },
38
+ "LICENSE": {
39
+ "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a",
40
+ "bytes": 11544
41
+ },
42
+ "NOTICE": {
43
+ "sha256": "39ce15275bd1433d18dd09066e9045529a3c7f1a8c3e761d3f318041167e9fed",
44
+ "bytes": 2664
45
+ },
46
+ "README.md": {
47
+ "sha256": "965607319a7721d4eb497128179a740b2c9ca9516761c028c44fa8c6b4d20733",
48
+ "bytes": 13399
49
+ },
50
+ "THIRD_PARTY_NOTICES.md": {
51
+ "sha256": "55ade7a70bb6bfcb8f0a8e9f4aef9222cde16c8bb44b6135aab6f929c706cbc5",
52
+ "bytes": 2146
53
+ },
54
+ "bf16/chat_template.jinja": {
55
+ "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80",
56
+ "bytes": 7755
57
+ },
58
+ "bf16/config.json": {
59
+ "sha256": "1dd31077b37706dc40e1bbddedf5eeba3e50e1d3c85d1fe34bfd5957a8d6ad14",
60
+ "bytes": 2000
61
+ },
62
+ "bf16/generation_config.json": {
63
+ "sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73",
64
+ "bytes": 116
65
+ },
66
+ "bf16/macjev_norms_fp32.safetensors": {
67
+ "sha256": "a30e1a72d02598b405978c0eff08c3bfca3b523fa0b3d4b4c80d2a185e0b8081",
68
+ "bytes": 420112
69
+ },
70
+ "bf16/model.safetensors": {
71
+ "sha256": "00629c8d17ea115c21c3fc0d6ddc5d4aa02887343a8ff3b3cf79ab197624d574",
72
+ "bytes": 3763691755
73
+ },
74
+ "bf16/model.safetensors.index.json": {
75
+ "sha256": "d49d69fd9257879080337533c6051a14ff4f561a79b9f70d8bdb74e47444a306",
76
+ "bytes": 28024
77
+ },
78
+ "bf16/tokenizer.json": {
79
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
80
+ "bytes": 19989325
81
+ },
82
+ "bf16/tokenizer_config.json": {
83
+ "sha256": "95c557768e6b88a7128befc7bfd3c7de50e5d51af9b8b33a9f4dee0e04f99679",
84
+ "bytes": 1161
85
+ },
86
+ "config.json": {
87
+ "sha256": "1dd31077b37706dc40e1bbddedf5eeba3e50e1d3c85d1fe34bfd5957a8d6ad14",
88
+ "bytes": 2000
89
+ },
90
+ "figures/banner.data.json": {
91
+ "sha256": "397cbfb222986f8d59a65143f2fec5a11170385c661a86c5e54d890a24d7d89e",
92
+ "bytes": 1109
93
+ },
94
+ "figures/banner.png": {
95
+ "sha256": "2623e4afad36239aaa007c34350d6f2fbacb33430ffd3c36ce2f7ee7db849b47",
96
+ "bytes": 447467
97
+ },
98
+ "figures/jevbench.data.json": {
99
+ "sha256": "508994afea8d8b1bc33e25c675621a9b3b55b01f900daa688c240c8356dffe3b",
100
+ "bytes": 2387
101
+ },
102
+ "figures/jevbench.png": {
103
+ "sha256": "c8542dba9ecc1bd2a4ea3bda844cbc3fbd661c40d6eb9254098b279a92a74ce8",
104
+ "bytes": 167843
105
+ },
106
+ "figures/jevbench.svg": {
107
+ "sha256": "13193f8d636ad5e6685cbc58863f27bced87c0ead94c252917839c7c6eed837e",
108
+ "bytes": 12280
109
+ },
110
+ "figures/zeroshot.data.json": {
111
+ "sha256": "3ad81c3dc926c988db5c339eec4991c2e2cce46faafb89db99cc47d6d8737538",
112
+ "bytes": 1841
113
+ },
114
+ "figures/zeroshot.png": {
115
+ "sha256": "d51f9a7bc46b88d8dcf5ba46f15d479c781302ca531208ff0838106da8aae755",
116
+ "bytes": 137714
117
+ },
118
+ "figures/zeroshot.svg": {
119
+ "sha256": "f3234ba4715b0d87e081ec9fc52c0b409010612f54db7aa39307a9eda3b7545f",
120
+ "bytes": 13874
121
+ },
122
+ "jev_style_decision_mlx.py": {
123
+ "sha256": "025daab9b374d740e743c519bd3aa3385791e08c81f4515b2ae525af57788030",
124
+ "bytes": 54020
125
+ },
126
+ "readout_config.json": {
127
+ "sha256": "4af0d578c9126d4eb1a545b6e576e0a49a967243a47b8e99e3555f081fdaa1de",
128
+ "bytes": 2031
129
+ },
130
+ "release_config.json": {
131
+ "sha256": "eddb00c6bf6682f3f62d0c69ffa3ef3820454abd8bea092c7ecddece55f5b551",
132
+ "bytes": 17829
133
+ },
134
+ "requirements.txt": {
135
+ "sha256": "2dd1d057db2c4f0d906e3665745e85d4647be032e1ffd2cc94cea2db2104c329",
136
+ "bytes": 479
137
+ },
138
+ "validation/SOURCES.json": {
139
+ "sha256": "fa8510c5c1f3afbdb01a3da6f26500cfa3117b8f4e26ceaf788fbf1eca9a5639",
140
+ "bytes": 3165
141
+ },
142
+ "validation/latency_2b.json": {
143
+ "sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
144
+ "bytes": 40294
145
+ },
146
+ "validation/parity/PREDECLARED_RELEASE_GATES_2B.md": {
147
+ "sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
148
+ "bytes": 2619
149
+ },
150
+ "validation/parity/compare_mlx_affine8-g64_base_vs_block.json": {
151
+ "sha256": "e1911f21ce73dd42d6c57e8349e2ec3221386a2b44ca461a3e50b9a69057ec34",
152
+ "bytes": 7697
153
+ },
154
+ "validation/parity/compare_mlx_affine8-g64_gate_vs_block.json": {
155
+ "sha256": "f0e2f44498df00945901c2a9bd5bff6f742d11ab243495c8089eaf0448fda782",
156
+ "bytes": 6494
157
+ },
158
+ "validation/parity/compare_mlx_bf16_base_vs_block.json": {
159
+ "sha256": "ca16e5281debaf1bc601b4a8c93cf6c1106f7b032f35c5833baa4fc8178def7c",
160
+ "bytes": 7408
161
+ },
162
+ "validation/parity/compare_mlx_bf16_gate_vs_block.json": {
163
+ "sha256": "7fa30fa414a4ffc12a7aa6371e77579845c3b816b90da3a813a70fec10ccd9fb",
164
+ "bytes": 6410
165
+ },
166
+ "validation/parity/cross_format_dp.json": {
167
+ "sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533",
168
+ "bytes": 1550
169
+ },
170
+ "validation/runtime/fixture_verify_8bit.json": {
171
+ "sha256": "305cc657c5ebd539ce92e3fbe7db36c594e40a0566d185b36dc858e338025d01",
172
+ "bytes": 3078
173
+ },
174
+ "validation/runtime/fixture_verify_bf16.json": {
175
+ "sha256": "87766f14dfdf9ac1d6d97692e523e3942329b4431615a46322992d1dbf315c63",
176
+ "bytes": 3077
177
+ },
178
+ "validation/runtime/render_verify.json": {
179
+ "sha256": "5333dd7e4083de7e80bd98b4da4d0965c2835782a37e836d55d0e09a26c4f8c1",
180
+ "bytes": 1550
181
+ },
182
+ "validation/runtime/reverify_mlx.json": {
183
+ "sha256": "1f6f62c8b66e8df2f738acfd96667ef746f9504328e603bdeaa848156ddf22bf",
184
+ "bytes": 963
185
+ }
186
+ },
187
+ "readme_hashed": true,
188
+ "readme_placeholder": false,
189
+ "note": "manifest.json hashes every file of the repo except itself, README.md, figures/ and validation/ included. The runtime --verify check (jev_style_decision_mlx.py) checks the top-level files (except README.md) and the selected precision folder (bf16/ or 8bit/), so either folder may be downloaded alone; README.md, figures/ and validation/ are not checked."
190
+ }
readout_config.json ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "macjev-readout-v2",
3
+ "model_name": "Jev-Style-2B-Decision-v3",
4
+ "readout": "verdict",
5
+ "template": "macjev-render-v2-long-options",
6
+ "layout": "sb",
7
+ "block_size": 2048,
8
+ "total_context_limit": 25600,
9
+ "question_option_limit": null,
10
+ "slot_tokens": {
11
+ "yes": {
12
+ "text": " yes",
13
+ "id": 9542
14
+ },
15
+ "no": {
16
+ "text": " no",
17
+ "id": 874
18
+ },
19
+ "verdict_slot": {
20
+ "text": " ->",
21
+ "id": 1411
22
+ }
23
+ },
24
+ "attention": "block-causal in the 6 full-attention layers: the input is cut into [start, stop) blocks of at most block_size tokens (state blocks; one short question block, or catalogue blocks + rubric blocks when question + options + slots exceed block_size); every block attends to all earlier tokens and to itself with no causal mask inside the block. Gated-DeltaNet layers are ordinary recurrent layers.",
25
+ "score": "per option k: logit[' yes'] - logit[' no'] at the k-th ' ->' slot, computed as h_slot . (w_yes - w_no) from the final normed hidden state and the tied embedding rows (float32)",
26
+ "probabilities": "softmax(scores / T) in canonical option order; T = temperatures.global (one global temperature)",
27
+ "tokenizer_sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523",
28
+ "budgets": {
29
+ "max_len": 25600,
30
+ "question_option_limit": null,
31
+ "note": "the complete input (state + question + options + readout) <= max_len tokens; there is no separate question/options cap; larger inputs raise InputBudgetError, nothing is truncated"
32
+ },
33
+ "temperatures": {
34
+ "version": "macjev-temperature-v2-global",
35
+ "global": 0.8278650620942867,
36
+ "groups": null,
37
+ "groups_note": "none: this model has one global temperature (no category / family / option-count temperatures)",
38
+ "fitted_on": "2,000 independent calibration rows (not dev, not test)",
39
+ "n_rows": 2000,
40
+ "fit_quality": {
41
+ "nll_before": 0.3002617012172898,
42
+ "nll_after": 0.2900580002426921,
43
+ "n": 2000
44
+ },
45
+ "selected_weights": "ema"
46
+ }
47
+ }
release_config.json ADDED
@@ -0,0 +1,422 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-release-v1",
3
+ "model_name": "Jev-Style-2B-Decision-v3",
4
+ "repo": "chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX",
5
+ "generation": "v3 (third generation of the Jev-Style decision series)",
6
+ "lineage": {
7
+ "v1": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision",
8
+ "v1_public_gguf": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-GGUF",
9
+ "v2": "chaoliangUNSW/Jev-Style-Qwen3.5-2B-Decision-v2",
10
+ "v3_0.8b": "chaoliangUNSW/Jev-Style-0.8B-Decision-v3",
11
+ "v3_2b": "chaoliangUNSW/Jev-Style-2B-Decision-v3"
12
+ },
13
+ "base_model": "Qwen/Qwen3.5-2B",
14
+ "base_model_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc",
15
+ "base_model_relation": "finetune",
16
+ "architecture": "Qwen3_5ForCausalLM (text only, 24 layers: 18 Gated DeltaNet + 6 full attention, hidden 2048, tied embeddings)",
17
+ "readout": "verdict",
18
+ "template": "macjev-render-v2-long-options",
19
+ "layout": "sb",
20
+ "attention": "block-causal (2,048-token blocks) in the full-attention layers; see readout_config.json -> attention",
21
+ "readout_config": "readout_config.json",
22
+ "budgets": {
23
+ "max_len": 25600,
24
+ "block_size": 2048,
25
+ "question_option_limit": null
26
+ },
27
+ "source": {
28
+ "checkpoint_sha256": {
29
+ "model-00001-of-00002.safetensors": "df592f869cb1232226d31491b783bfa9406ffb3c795eae44a3b7452e1044a8fb",
30
+ "model-00002-of-00002.safetensors": "c6ff77a32a8cf36767e51c1282bb345a873db6509847f6c75baadd8fb17eeb70"
31
+ },
32
+ "selected_weights": "ema",
33
+ "mlx": "0.32.2",
34
+ "mlx_lm": "0.31.3",
35
+ "exports_manifest": {
36
+ "file": "runs/macjev/release_2b/exports_manifest.json",
37
+ "sha256": "11c549e479c7022c960c546730d6a3c42957c1c40b1b8901a183a893917d96a5",
38
+ "note": "sha256 / bytes of the exported GGUF and MLX files (release_2b/build_exports.py)"
39
+ },
40
+ "candidate_sha256sums": {
41
+ "file": "runs/macjev/candidate_2b/hf-candidate/SHA256SUMS.json",
42
+ "sha256": "7549ed3fe6239b2862998798d8da1457fdced25cd424aa4442557ae852d6e696"
43
+ }
44
+ },
45
+ "calibration": {
46
+ "version": "macjev-temperature-v2-global",
47
+ "global_T": 0.8278650620942867,
48
+ "groups": null,
49
+ "n_rows": 2000,
50
+ "fit_quality": {
51
+ "nll_before": 0.3002617012172898,
52
+ "nll_after": 0.2900580002426921,
53
+ "n": 2000
54
+ },
55
+ "fitted_on": "2,000 independent calibration rows (not dev, not test)"
56
+ },
57
+ "related_repos": {
58
+ "main": "chaoliangUNSW/Jev-Style-2B-Decision-v3",
59
+ "gguf": "chaoliangUNSW/Jev-Style-2B-Decision-v3-GGUF",
60
+ "mlx": "chaoliangUNSW/Jev-Style-2B-Decision-v3-MLX"
61
+ },
62
+ "tested_with": {
63
+ "python": "3.12",
64
+ "mlx": "0.32.2",
65
+ "mlx-lm": "0.31.3",
66
+ "tokenizers": "0.23.2",
67
+ "numpy": "2.5.3"
68
+ },
69
+ "weights": {
70
+ "bf16/model.safetensors": "bfloat16",
71
+ "8bit/model.safetensors": "mlx affine 8-bit, group 64",
72
+ "*/macjev_norms_fp32.safetensors": "exact FP32 (1 + w) of the 61 shifted Qwen3.5 RMSNorms (format macjev-norms-fp32-v1); required by the runtime"
73
+ },
74
+ "runtime": {
75
+ "script": "jev_style_decision_mlx.py",
76
+ "class": "JevStyleDecisionMLX",
77
+ "precision_argument": "--precision bf16|8bit (Python: precision=...)",
78
+ "default_precision": "bf16",
79
+ "mlx_cache_limit_gib": 2.0,
80
+ "mlx_lm_required": "0.31.3 (exact; the runtime checks the sha256 of the mlx-lm source it patches and refuses other versions)",
81
+ "mlx_lm_corrections": {
82
+ "gdn_qk_l2norm": "Gated-DeltaNet q/k normalised with l2norm eps 1e-6 (HF / fla / llama.cpp) instead of mlx-lm's rms_norm eps 1e-6 (= l2norm eps Dk*1e-6); per-instance class swap",
83
+ "rmsnorm_fp32_shift": "Qwen3.5 (1 + w) RMSNorm weights taken in FP32 from macjev_norms_fp32.safetensors instead of the bf16-rounded (1 + w) stored by mlx-lm's converter; norms computed in FP32 and cast back"
84
+ },
85
+ "state_reuse": "score_many(state, questions) computes the state blocks once and runs every question on a copy of that cache; the last state is kept, so consecutive calls / JSONL rows with an identical state reuse it",
86
+ "shared_files": "readout_config.json (readout + calibration temperature), release_config.json and manifest.json are shared by both precisions; each precision folder has its own config.json, tokenizer, weights and norm sidecar",
87
+ "precisions": {
88
+ "bf16": {
89
+ "folder": "bf16/",
90
+ "weights": "bfloat16",
91
+ "mlx_compute_dtype": null,
92
+ "mlx_compute_dtype_note": "native activations (bf16 residual stream; norms and readout in FP32), the setting the release scoring used"
93
+ },
94
+ "8bit": {
95
+ "folder": "8bit/",
96
+ "weights": "mlx affine 8-bit, group 64",
97
+ "mlx_compute_dtype": null,
98
+ "mlx_compute_dtype_note": "native activations, the setting the release scoring used"
99
+ }
100
+ },
101
+ "removed_vs_0.8b_runtime": {
102
+ "--category": "no group temperatures: one global temperature",
103
+ "--head-max": "no question/options cap: only the 25,600-token total budget; long question+options use the catalogue-overflow layout",
104
+ "option chunking": "not needed: all options are always scored in one input"
105
+ }
106
+ },
107
+ "format_gates": {
108
+ "predeclared": {
109
+ "file": "runs/macjev/runtime_v2_dev/PREDECLARED_RELEASE_GATES_2B.md",
110
+ "sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
111
+ "gates": "F16/bf16 top-1 >= 0.99; 8-bit top-1 >= 0.98; |dNLL| <= 0.02; accuracy drop Q8_0 / MLX-8bit <= 0.3 pp, Q4_K_M <= 1.0 pp (4-bit top-1 / NLL report-only); no non-finite scores"
112
+ },
113
+ "reference": {
114
+ "what": "HF transformers FP32 on CPU, exact v2 block attention, released bf16 checkpoint (runs/macjev/candidate_2b/hf-candidate)",
115
+ "temperature": 0.8278650620942867
116
+ },
117
+ "fixtures": {
118
+ "gate": {
119
+ "what": "1,000 real dev rows (<= 4,096 tokens); decides the release gates",
120
+ "file": "runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl",
121
+ "sha256": "1c03b8c570b988b0c83ca1f79e7fd37e6196239a0836f348300c3f23ae667b68"
122
+ },
123
+ "long": {
124
+ "what": "35 requests / 43 questions up to 25,600 tokens incl. catalogue overflow and K <= 151; accuracy gate unresolvable on 43 questions (verdict INCONCLUSIVE means only that), top-1 / dNLL reported",
125
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
126
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
127
+ }
128
+ },
129
+ "cross_format_dp": {
130
+ "file": "runs/macjev/release_2b/parity/cross_format_dp.json",
131
+ "sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533"
132
+ },
133
+ "formats": {
134
+ "mlx-bf16": {
135
+ "gate_fixture": {
136
+ "n": 1000,
137
+ "top1_agreement": 0.997,
138
+ "top1_flips": 3,
139
+ "max_abs_dscore": 0.13961565494537354,
140
+ "mean_abs_dscore": 0.012822698334419146,
141
+ "dnll": 0.0001265220422789204,
142
+ "nll_ref": 0.4436679457518203,
143
+ "nll_cand": 0.4437944677940992,
144
+ "acc_ref": 0.808,
145
+ "acc_cand": 0.805,
146
+ "acc_drop_pp": 0.30000000000000027,
147
+ "max_abs_dp": 0.03549758339084519,
148
+ "mean_max_abs_dp": 0.0025865935031305484,
149
+ "gates": {
150
+ "coverage_finite_tokens": {
151
+ "pass": true,
152
+ "problems": {},
153
+ "reference_missing_requests": 0
154
+ },
155
+ "top1": {
156
+ "threshold": 0.99,
157
+ "value": 0.997,
158
+ "pass": true
159
+ },
160
+ "dnll": {
161
+ "threshold": 0.02,
162
+ "value": 0.0001265220422789204,
163
+ "pass": true
164
+ }
165
+ },
166
+ "verdict": "PASS",
167
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
168
+ "source": {
169
+ "file": "runs/macjev/release_2b/parity/compare_mlx_bf16_gate_vs_block.json",
170
+ "sha256": "7fa30fa414a4ffc12a7aa6371e77579845c3b816b90da3a813a70fec10ccd9fb"
171
+ }
172
+ },
173
+ "long_fixture": {
174
+ "n": 43,
175
+ "top1_agreement": 1.0,
176
+ "top1_flips": 0,
177
+ "max_abs_dscore": 0.08836114406585693,
178
+ "mean_abs_dscore": 0.01365492056503762,
179
+ "dnll": -0.0026943261000204055,
180
+ "nll_ref": 0.670433307194272,
181
+ "nll_cand": 0.6677389810942516,
182
+ "acc_ref": 0.7209302325581395,
183
+ "acc_cand": 0.7209302325581395,
184
+ "acc_drop_pp": 0.0,
185
+ "max_abs_dp": 0.010765431767972844,
186
+ "mean_max_abs_dp": 0.003890125022245109,
187
+ "gates": {
188
+ "coverage_finite_tokens": {
189
+ "pass": true,
190
+ "problems": {},
191
+ "reference_missing_requests": 0
192
+ },
193
+ "top1": {
194
+ "threshold": 0.99,
195
+ "value": 1.0,
196
+ "pass": true
197
+ },
198
+ "dnll": {
199
+ "threshold": 0.02,
200
+ "value": -0.0026943261000204055,
201
+ "pass": true
202
+ }
203
+ },
204
+ "verdict": "PASS",
205
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
206
+ "source": {
207
+ "file": "runs/macjev/release_2b/parity/compare_mlx_bf16_base_vs_block.json",
208
+ "sha256": "ca16e5281debaf1bc601b4a8c93cf6c1106f7b032f35c5833baa4fc8178def7c"
209
+ }
210
+ },
211
+ "release_verdict": "PASS"
212
+ },
213
+ "mlx-8bit": {
214
+ "gate_fixture": {
215
+ "n": 1000,
216
+ "top1_agreement": 0.996,
217
+ "top1_flips": 4,
218
+ "max_abs_dscore": 0.3184394836425781,
219
+ "mean_abs_dscore": 0.016680597170951345,
220
+ "dnll": 0.0005035682350758575,
221
+ "nll_ref": 0.4436679457518203,
222
+ "nll_cand": 0.44417151398689614,
223
+ "acc_ref": 0.808,
224
+ "acc_cand": 0.808,
225
+ "acc_drop_pp": 0.0,
226
+ "max_abs_dp": 0.1623697296878569,
227
+ "mean_max_abs_dp": 0.0039011049015487734,
228
+ "gates": {
229
+ "coverage_finite_tokens": {
230
+ "pass": true,
231
+ "problems": {},
232
+ "reference_missing_requests": 0
233
+ },
234
+ "top1": {
235
+ "threshold": 0.98,
236
+ "value": 0.996,
237
+ "pass": true
238
+ },
239
+ "dnll": {
240
+ "threshold": 0.02,
241
+ "value": 0.0005035682350758575,
242
+ "pass": true
243
+ },
244
+ "acc_drop_pp": {
245
+ "threshold": 0.3,
246
+ "value": 0.0,
247
+ "resolution_pp": 0.1,
248
+ "pass": true,
249
+ "note": null
250
+ }
251
+ },
252
+ "verdict": "PASS",
253
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
254
+ "source": {
255
+ "file": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_gate_vs_block.json",
256
+ "sha256": "f0e2f44498df00945901c2a9bd5bff6f742d11ab243495c8089eaf0448fda782"
257
+ }
258
+ },
259
+ "long_fixture": {
260
+ "n": 43,
261
+ "top1_agreement": 1.0,
262
+ "top1_flips": 0,
263
+ "max_abs_dscore": 0.4653351306915283,
264
+ "mean_abs_dscore": 0.023077200647273945,
265
+ "dnll": 0.01638063705636561,
266
+ "nll_ref": 0.670433307194272,
267
+ "nll_cand": 0.6868139442506376,
268
+ "acc_ref": 0.7209302325581395,
269
+ "acc_cand": 0.7209302325581395,
270
+ "acc_drop_pp": 0.0,
271
+ "max_abs_dp": 0.04628699142360499,
272
+ "mean_max_abs_dp": 0.007026009064181663,
273
+ "gates": {
274
+ "coverage_finite_tokens": {
275
+ "pass": true,
276
+ "problems": {},
277
+ "reference_missing_requests": 0
278
+ },
279
+ "top1": {
280
+ "threshold": 0.98,
281
+ "value": 1.0,
282
+ "pass": true
283
+ },
284
+ "dnll": {
285
+ "threshold": 0.02,
286
+ "value": 0.01638063705636561,
287
+ "pass": true
288
+ },
289
+ "acc_drop_pp": {
290
+ "threshold": 0.3,
291
+ "value": 0.0,
292
+ "resolution_pp": 2.3255813953488373,
293
+ "pass": null,
294
+ "note": "UNRESOLVED: 43 gold questions -> one flip = 2.33pp > 0.3pp; this fixture cannot resolve the gate (use the gate fixture with >= 334 rows; statistical power needs far more)"
295
+ }
296
+ },
297
+ "verdict": "INCONCLUSIVE",
298
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
299
+ "source": {
300
+ "file": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_base_vs_block.json",
301
+ "sha256": "e1911f21ce73dd42d6c57e8349e2ec3221386a2b44ca461a3e50b9a69057ec34"
302
+ }
303
+ },
304
+ "release_verdict": "PASS"
305
+ }
306
+ }
307
+ },
308
+ "decision_index": "requested from the maintainer after release (not run by us)",
309
+ "runtime_parity": {
310
+ "protocol": "the staged runtime scores the long fixture from text/ids through its own renderer; raw scores compared with the development MLX engine that produced the release parity (release_2b/parity/pred_mlx_*_base.jsonl)",
311
+ "fixture": {
312
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
313
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
314
+ },
315
+ "bf16": {
316
+ "precision": "bf16",
317
+ "dev_seconds": 415.6,
318
+ "default_T": 0.8278650620942867,
319
+ "fixture_requests": 35,
320
+ "fixture_questions": 43,
321
+ "fixture_tokens": 214700,
322
+ "overflow_q": 5,
323
+ "max_abs_dscore": 0.0,
324
+ "max_abs_dprob_vs_softmax_dev_over_T": 0.0,
325
+ "text_e2e_max_abs_dscore_vs_dev": 0.0,
326
+ "text_ids_equal_dev": true,
327
+ "score_many_vs_separate_fresh_decide_max_abs": 0.0,
328
+ "order_invariance_max_abs": 0.0,
329
+ "runtime_seconds": 293.1,
330
+ "source": {
331
+ "file": "runs/macjev/runtime_v2_dev/verify_release_mlx/fixture_verify_bf16.json",
332
+ "sha256": "87766f14dfdf9ac1d6d97692e523e3942329b4431615a46322992d1dbf315c63"
333
+ }
334
+ },
335
+ "8bit": {
336
+ "precision": "8bit",
337
+ "dev_seconds": 1324.8,
338
+ "default_T": 0.8278650620942867,
339
+ "fixture_requests": 35,
340
+ "fixture_questions": 43,
341
+ "fixture_tokens": 214700,
342
+ "overflow_q": 5,
343
+ "max_abs_dscore": 0.0,
344
+ "max_abs_dprob_vs_softmax_dev_over_T": 0.0,
345
+ "text_e2e_max_abs_dscore_vs_dev": 0.0,
346
+ "text_ids_equal_dev": true,
347
+ "score_many_vs_separate_fresh_decide_max_abs": 0.0,
348
+ "order_invariance_max_abs": 0.0,
349
+ "runtime_seconds": 286.5,
350
+ "source": {
351
+ "file": "runs/macjev/runtime_v2_dev/verify_release_mlx/fixture_verify_8bit.json",
352
+ "sha256": "305cc657c5ebd539ce92e3fbe7db36c594e40a0566d185b36dc858e338025d01"
353
+ }
354
+ },
355
+ "render_token_identity": {
356
+ "requests": 463,
357
+ "questions": 756,
358
+ "ok_identical": 743,
359
+ "budget_both": 6,
360
+ "error_both": 6,
361
+ "mismatch_count": 1,
362
+ "source": {
363
+ "file": "runs/macjev/runtime_v2_dev/verify_release_mlx/render_verify.json",
364
+ "sha256": "5333dd7e4083de7e80bd98b4da4d0965c2835782a37e836d55d0e09a26c4f8c1"
365
+ }
366
+ },
367
+ "runtime_file": {
368
+ "file": "jev_style_decision_mlx.py",
369
+ "sha256_now": "025daab9b374d740e743c519bd3aa3385791e08c81f4515b2ae525af57788030",
370
+ "mtime_unix": 1790448336.812169
371
+ },
372
+ "original_records_predate_last_runtime_edit": true,
373
+ "reverified_on_current_runtime": true,
374
+ "recorded_before_last_runtime_edit": false,
375
+ "note": "the records above were written before the last edit of the runtime file; the runtime as staged now (sha256 runtime_file.sha256_now) was re-verified on the long fixture, see 'reverified' (reverified.runtime_sha256 == runtime_file.sha256_now).",
376
+ "reverified": {
377
+ "runtime_file": "jev_style_decision_mlx.py",
378
+ "runtime_sha256": "025daab9b374d740e743c519bd3aa3385791e08c81f4515b2ae525af57788030",
379
+ "shared_core_sha256": "56deae095206fc59b9a30d8626ac5d27cacddcd94d7c7f08f6f4bc53d326a002",
380
+ "what": "MLX bf16 (loaded from the repository root, precision bf16), runtime defaults",
381
+ "loaded_from": "runs/macjev/hf_staging/Jev-Style-2B-Decision-v3-MLX",
382
+ "verify_manifest": true,
383
+ "compared_with": {
384
+ "file": "runs/macjev/release_2b/parity/pred_mlx_bf16_base.jsonl",
385
+ "sha256": "c42232870ab8c5410b4f45dc85729249d467b9c0bd5d24213d085e4479a3a60e"
386
+ },
387
+ "fixture": {
388
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
389
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
390
+ },
391
+ "questions": 43,
392
+ "skipped_questions": 0,
393
+ "max_abs_score_diff": 0.0,
394
+ "top1_same": "43/43",
395
+ "token_or_overflow_mismatches": [],
396
+ "seconds": 137.2,
397
+ "finished_unix": 1790449492.3913028,
398
+ "source": {
399
+ "file": "runs/macjev/hf_staging/_2b_tools/reverify_mlx.json",
400
+ "sha256": "1f6f62c8b66e8df2f738acfd96667ef746f9504328e603bdeaa848156ddf22bf"
401
+ }
402
+ }
403
+ },
404
+ "latency": {
405
+ "file": "validation/latency_2b.json",
406
+ "sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
407
+ "machine": {
408
+ "chip": "Apple M1 Max",
409
+ "memory_bytes": 68719476736,
410
+ "macos": "15.7.5",
411
+ "python": "3.12.13"
412
+ },
413
+ "backends_shown_on_card": [
414
+ "mlx-bf16",
415
+ "mlx-8bit"
416
+ ],
417
+ "rows": 12,
418
+ "rows_in_file": 36,
419
+ "card_generator": "runs/macjev/release_2b/cards/_build/latency_tables.py"
420
+ },
421
+ "root_config_json": "copy of bf16/config.json at the repository root so the Hub counts downloads; the runtime treats the root as the repository folder (it has bf16/ and 8bit/ but no model.safetensors) and loads bf16/ or 8bit/"
422
+ }
requirements.txt ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ # tested with: python 3.12, mlx 0.32.2, mlx-lm 0.31.3, transformers 5.17.0, tokenizers 0.23.2, numpy 2.5.3 (Apple silicon, macOS 15)
2
+ # mlx-lm is pinned: the runtime corrects two Qwen3.5 numerics inside mlx-lm (Gated-DeltaNet q/k l2norm eps,
3
+ # FP32 (1 + w) RMSNorm weights) and checks the sha256 of the mlx-lm source it patches; other versions are refused.
4
+ mlx==0.32.2
5
+ mlx-lm==0.31.3
6
+ transformers==5.17.0 # imported by mlx-lm when the model loads
7
+ tokenizers==0.23.2
8
+ numpy==2.5.3
validation/SOURCES.json ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "note": "copies of the verification / benchmark records behind release_config.json and the model card; local absolute paths replaced by repository-relative ones (local_paths_scrubbed); source_sha256 = the unmodified file (equal to the published copy when local_paths_scrubbed is false)",
3
+ "files": {
4
+ "parity/compare_mlx_bf16_gate_vs_block.json": {
5
+ "source": "runs/macjev/release_2b/parity/compare_mlx_bf16_gate_vs_block.json",
6
+ "source_sha256": "7fa30fa414a4ffc12a7aa6371e77579845c3b816b90da3a813a70fec10ccd9fb",
7
+ "local_paths_scrubbed": false
8
+ },
9
+ "parity/compare_mlx_bf16_base_vs_block.json": {
10
+ "source": "runs/macjev/release_2b/parity/compare_mlx_bf16_base_vs_block.json",
11
+ "source_sha256": "ca16e5281debaf1bc601b4a8c93cf6c1106f7b032f35c5833baa4fc8178def7c",
12
+ "local_paths_scrubbed": false
13
+ },
14
+ "parity/compare_mlx_affine8-g64_gate_vs_block.json": {
15
+ "source": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_gate_vs_block.json",
16
+ "source_sha256": "f0e2f44498df00945901c2a9bd5bff6f742d11ab243495c8089eaf0448fda782",
17
+ "local_paths_scrubbed": false
18
+ },
19
+ "parity/compare_mlx_affine8-g64_base_vs_block.json": {
20
+ "source": "runs/macjev/release_2b/parity/compare_mlx_affine8-g64_base_vs_block.json",
21
+ "source_sha256": "e1911f21ce73dd42d6c57e8349e2ec3221386a2b44ca461a3e50b9a69057ec34",
22
+ "local_paths_scrubbed": false
23
+ },
24
+ "parity/cross_format_dp.json": {
25
+ "source": "runs/macjev/release_2b/parity/cross_format_dp.json",
26
+ "source_sha256": "c7332df0f02244416b01e49ecb037c76d4bf2643a2b42505ff1dfe896879e533",
27
+ "local_paths_scrubbed": false
28
+ },
29
+ "parity/PREDECLARED_RELEASE_GATES_2B.md": {
30
+ "source": "runs/macjev/runtime_v2_dev/PREDECLARED_RELEASE_GATES_2B.md",
31
+ "source_sha256": "783aec77300779582911cdc57c71cf8b101b0cd5ca040073f7e1fa3ac20af08e",
32
+ "local_paths_scrubbed": false
33
+ },
34
+ "runtime/fixture_verify_bf16.json": {
35
+ "source": "runs/macjev/runtime_v2_dev/verify_release_mlx/fixture_verify_bf16.json",
36
+ "source_sha256": "87766f14dfdf9ac1d6d97692e523e3942329b4431615a46322992d1dbf315c63",
37
+ "local_paths_scrubbed": false
38
+ },
39
+ "runtime/fixture_verify_8bit.json": {
40
+ "source": "runs/macjev/runtime_v2_dev/verify_release_mlx/fixture_verify_8bit.json",
41
+ "source_sha256": "305cc657c5ebd539ce92e3fbe7db36c594e40a0566d185b36dc858e338025d01",
42
+ "local_paths_scrubbed": false
43
+ },
44
+ "runtime/render_verify.json": {
45
+ "source": "runs/macjev/runtime_v2_dev/verify_release_mlx/render_verify.json",
46
+ "source_sha256": "5333dd7e4083de7e80bd98b4da4d0965c2835782a37e836d55d0e09a26c4f8c1",
47
+ "local_paths_scrubbed": false
48
+ },
49
+ "runtime/reverify_mlx.json": {
50
+ "source": "runs/macjev/hf_staging/_2b_tools/reverify_mlx.json",
51
+ "source_sha256": "1f6f62c8b66e8df2f738acfd96667ef746f9504328e603bdeaa848156ddf22bf",
52
+ "local_paths_scrubbed": false
53
+ },
54
+ "latency_2b.json": {
55
+ "source": "runs/macjev/release_2b/latency_2b.json",
56
+ "source_sha256": "458d00e0c81b3165072b8547278e6039b9ae0d45d54b7421008ef24c1d24ca3d",
57
+ "local_paths_scrubbed": false
58
+ }
59
+ }
60
+ }
validation/latency_2b.json ADDED
@@ -0,0 +1,1568 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "jev-style-latency-v1",
3
+ "model": "Jev-Style-2B-Decision-v3",
4
+ "machine": {
5
+ "chip": "Apple M1 Max",
6
+ "memory_bytes": 68719476736,
7
+ "macos": "15.7.5",
8
+ "python": "3.12.13"
9
+ },
10
+ "shared_machine_note": "measured while other agents' jobs ran on the same Mac (among them a CPU-heavy PyTorch parity job using ~3.5 cores and ~16 GB); load averages at the end of each run are recorded per row. Treat the numbers as indicative, not as a clean benchmark.",
11
+ "protocol": {
12
+ "runtimes": "the staged runtimes of the three repos (jev_style_decision_gguf.py + jev-score-v2 built with build_jev_score.sh against llama.cpp 441df11f, Metal, all layers on the GPU; jev_style_decision_mlx.py, mlx 0.32.2 / mlx-lm 0.31.3; jev_style_decision.py on MPS, float32)",
13
+ "weights": "trained release weights: release_2b/gguf/model-*.gguf (tensor data identical to the named repo files), release_2b/mlx/{bf16,affine8-g64}, candidate_2b/hf-candidate (torch)",
14
+ "state": "plain text (repository documentation + source code, English) cut to target-150 tokens; the total input per question is recorded in input_tokens_per_question",
15
+ "questions": "n_questions=1: one 4-option choice question (decide); n_questions=10: 10 mixed questions (4 choice, 4 true/false, 2 score) about the same state in one score_many call",
16
+ "one_process_per_row": true,
17
+ "cold_s": "first scoring call after loading (state + questions; includes GPU warm-up)",
18
+ "warm_median_s": "median of 3 further calls, cached state dropped before each (state recomputed)",
19
+ "state_cached_median_s": "median of 3 calls with the state already computed (only the question blocks run)",
20
+ "load_s": "runtime construction (weights from the OS file cache in most rows)",
21
+ "wall_time": "time.perf_counter around decide()/score_many(), incl. tokenisation and rendering",
22
+ "peak_rss_bytes": "ru_maxrss of the Python process (and of the jev-score-v2 child for GGUF); includes memory-mapped weight pages",
23
+ "peak_phys_footprint_bytes": "macOS lifetime-max physical footprint (proc_pid_rusage v4): memory the process owns, incl. Metal / MLX buffers it allocates. It does NOT count clean memory-mapped file pages, so for GGUF (jev-score-v2 maps the .gguf file) it excludes the weights and is not a memory requirement; use peak_rss_bytes.jev_score_v2 (which includes the mapped weight pages) as the upper bound for GGUF"
24
+ },
25
+ "reruns": {
26
+ "gguf_and_mlx_rows": "all 18 GGUF rows and all 12 MLX rows were re-measured on 2026-09-27 03:11-03:28 (fix round, same script, same state text, same answers) because an independent re-run of the first session's GGUF Q8_0 / Q4_K_M 10-question rows was ~2.4x faster (contention during the first session). The first-session rows are kept in latency/runs_superseded_2026-09-27/ and are not used here. The 6 torch MPS rows are from the first session (not re-measured).",
27
+ "superseded_dir": "latency/runs_superseded_2026-09-27/"
28
+ },
29
+ "complete": true,
30
+ "missing_runs": [],
31
+ "rows": [
32
+ {
33
+ "backend": "gguf-f16",
34
+ "n_questions": 1,
35
+ "state_tokens": 878,
36
+ "input_tokens_per_question": [
37
+ 943
38
+ ],
39
+ "load_s": 2.6054,
40
+ "cold_s": 0.5624,
41
+ "warm_median_s": 0.5242,
42
+ "warm_s": [
43
+ 0.5203,
44
+ 0.5345,
45
+ 0.5242
46
+ ],
47
+ "state_cached_median_s": 0.0653,
48
+ "peak_rss_bytes": {
49
+ "python": 352616448,
50
+ "jev_score_v2": 4451008512
51
+ },
52
+ "peak_phys_footprint_bytes": {
53
+ "python": 252413696,
54
+ "jev_score_v2": 660384576
55
+ },
56
+ "results_identical_cold_vs_state_cached": true,
57
+ "loadavg_at_end": [
58
+ 9.59,
59
+ 9.9,
60
+ 10.98
61
+ ],
62
+ "measured_unix": 1790442718.8146281,
63
+ "settings": {
64
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
65
+ "n_gpu_layers": 999,
66
+ "flash_attn": "default (llama.cpp auto on Metal)",
67
+ "gguf": "release_2b/gguf/model-f16.gguf"
68
+ },
69
+ "raw": "latency/runs/gguf-f16_1024_1q.json"
70
+ },
71
+ {
72
+ "backend": "gguf-f16",
73
+ "n_questions": 10,
74
+ "state_tokens": 878,
75
+ "input_tokens_per_question": [
76
+ 943,
77
+ 918,
78
+ 925,
79
+ 911,
80
+ 918,
81
+ 921,
82
+ 943,
83
+ 917,
84
+ 912,
85
+ 919
86
+ ],
87
+ "load_s": 1.1941,
88
+ "cold_s": 0.994,
89
+ "warm_median_s": 0.968,
90
+ "warm_s": [
91
+ 0.968,
92
+ 0.9464,
93
+ 0.9713
94
+ ],
95
+ "state_cached_median_s": 0.4965,
96
+ "peak_rss_bytes": {
97
+ "python": 328564736,
98
+ "jev_score_v2": 4516610048
99
+ },
100
+ "peak_phys_footprint_bytes": {
101
+ "python": 254527296,
102
+ "jev_score_v2": 676899648
103
+ },
104
+ "results_identical_cold_vs_state_cached": true,
105
+ "loadavg_at_end": [
106
+ 10.0,
107
+ 9.97,
108
+ 10.99
109
+ ],
110
+ "measured_unix": 1790442726.19508,
111
+ "settings": {
112
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
113
+ "n_gpu_layers": 999,
114
+ "flash_attn": "default (llama.cpp auto on Metal)",
115
+ "gguf": "release_2b/gguf/model-f16.gguf"
116
+ },
117
+ "raw": "latency/runs/gguf-f16_1024_10q.json"
118
+ },
119
+ {
120
+ "backend": "gguf-f16",
121
+ "n_questions": 1,
122
+ "state_tokens": 3950,
123
+ "input_tokens_per_question": [
124
+ 4015
125
+ ],
126
+ "load_s": 1.1659,
127
+ "cold_s": 2.2122,
128
+ "warm_median_s": 2.1753,
129
+ "warm_s": [
130
+ 2.1909,
131
+ 2.1753,
132
+ 2.1733
133
+ ],
134
+ "state_cached_median_s": 0.078,
135
+ "peak_rss_bytes": {
136
+ "python": 354697216,
137
+ "jev_score_v2": 4568334336
138
+ },
139
+ "peak_phys_footprint_bytes": {
140
+ "python": 257705600,
141
+ "jev_score_v2": 722676736
142
+ },
143
+ "results_identical_cold_vs_state_cached": true,
144
+ "loadavg_at_end": [
145
+ 9.01,
146
+ 9.76,
147
+ 10.9
148
+ ],
149
+ "measured_unix": 1790442737.183161,
150
+ "settings": {
151
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
152
+ "n_gpu_layers": 999,
153
+ "flash_attn": "default (llama.cpp auto on Metal)",
154
+ "gguf": "release_2b/gguf/model-f16.gguf"
155
+ },
156
+ "raw": "latency/runs/gguf-f16_4096_1q.json"
157
+ },
158
+ {
159
+ "backend": "gguf-f16",
160
+ "n_questions": 10,
161
+ "state_tokens": 3950,
162
+ "input_tokens_per_question": [
163
+ 4015,
164
+ 3990,
165
+ 3997,
166
+ 3983,
167
+ 3990,
168
+ 3993,
169
+ 4015,
170
+ 3989,
171
+ 3984,
172
+ 3991
173
+ ],
174
+ "load_s": 1.2112,
175
+ "cold_s": 2.6904,
176
+ "warm_median_s": 2.6583,
177
+ "warm_s": [
178
+ 2.6583,
179
+ 2.6665,
180
+ 2.6378
181
+ ],
182
+ "state_cached_median_s": 0.5467,
183
+ "peak_rss_bytes": {
184
+ "python": 331694080,
185
+ "jev_score_v2": 4564172800
186
+ },
187
+ "peak_phys_footprint_bytes": {
188
+ "python": 248334016,
189
+ "jev_score_v2": 720382976
190
+ },
191
+ "results_identical_cold_vs_state_cached": true,
192
+ "loadavg_at_end": [
193
+ 8.62,
194
+ 9.62,
195
+ 10.83
196
+ ],
197
+ "measured_unix": 1790442751.5065348,
198
+ "settings": {
199
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
200
+ "n_gpu_layers": 999,
201
+ "flash_attn": "default (llama.cpp auto on Metal)",
202
+ "gguf": "release_2b/gguf/model-f16.gguf"
203
+ },
204
+ "raw": "latency/runs/gguf-f16_4096_10q.json"
205
+ },
206
+ {
207
+ "backend": "gguf-f16",
208
+ "n_questions": 1,
209
+ "state_tokens": 24436,
210
+ "input_tokens_per_question": [
211
+ 24501
212
+ ],
213
+ "load_s": 1.1784,
214
+ "cold_s": 16.2299,
215
+ "warm_median_s": 16.1651,
216
+ "warm_s": [
217
+ 16.1645,
218
+ 16.1651,
219
+ 16.2375
220
+ ],
221
+ "state_cached_median_s": 0.1677,
222
+ "peak_rss_bytes": {
223
+ "python": 349683712,
224
+ "jev_score_v2": 4734812160
225
+ },
226
+ "peak_phys_footprint_bytes": {
227
+ "python": 239191744,
228
+ "jev_score_v2": 882732352
229
+ },
230
+ "results_identical_cold_vs_state_cached": true,
231
+ "loadavg_at_end": [
232
+ 9.74,
233
+ 9.66,
234
+ 10.75
235
+ ],
236
+ "measured_unix": 1790442818.798799,
237
+ "settings": {
238
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
239
+ "n_gpu_layers": 999,
240
+ "flash_attn": "default (llama.cpp auto on Metal)",
241
+ "gguf": "release_2b/gguf/model-f16.gguf"
242
+ },
243
+ "raw": "latency/runs/gguf-f16_24576_1q.json"
244
+ },
245
+ {
246
+ "backend": "gguf-f16",
247
+ "n_questions": 10,
248
+ "state_tokens": 24436,
249
+ "input_tokens_per_question": [
250
+ 24501,
251
+ 24476,
252
+ 24483,
253
+ 24469,
254
+ 24476,
255
+ 24479,
256
+ 24501,
257
+ 24475,
258
+ 24470,
259
+ 24477
260
+ ],
261
+ "load_s": 1.225,
262
+ "cold_s": 16.9884,
263
+ "warm_median_s": 16.8115,
264
+ "warm_s": [
265
+ 16.8115,
266
+ 16.675,
267
+ 16.8156
268
+ ],
269
+ "state_cached_median_s": 0.8643,
270
+ "peak_rss_bytes": {
271
+ "python": 367919104,
272
+ "jev_score_v2": 4738170880
273
+ },
274
+ "peak_phys_footprint_bytes": {
275
+ "python": 242992832,
276
+ "jev_score_v2": 887975168
277
+ },
278
+ "results_identical_cold_vs_state_cached": true,
279
+ "loadavg_at_end": [
280
+ 8.02,
281
+ 9.14,
282
+ 10.46
283
+ ],
284
+ "measured_unix": 1790442890.769493,
285
+ "settings": {
286
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
287
+ "n_gpu_layers": 999,
288
+ "flash_attn": "default (llama.cpp auto on Metal)",
289
+ "gguf": "release_2b/gguf/model-f16.gguf"
290
+ },
291
+ "raw": "latency/runs/gguf-f16_24576_10q.json"
292
+ },
293
+ {
294
+ "backend": "gguf-q8_0",
295
+ "n_questions": 1,
296
+ "state_tokens": 878,
297
+ "input_tokens_per_question": [
298
+ 943
299
+ ],
300
+ "load_s": 1.7934,
301
+ "cold_s": 0.6136,
302
+ "warm_median_s": 0.5663,
303
+ "warm_s": [
304
+ 0.5678,
305
+ 0.5663,
306
+ 0.5663
307
+ ],
308
+ "state_cached_median_s": 0.0682,
309
+ "peak_rss_bytes": {
310
+ "python": 328974336,
311
+ "jev_score_v2": 2721579008
312
+ },
313
+ "peak_phys_footprint_bytes": {
314
+ "python": 248874624,
315
+ "jev_score_v2": 668163520
316
+ },
317
+ "results_identical_cold_vs_state_cached": true,
318
+ "loadavg_at_end": [
319
+ 7.69,
320
+ 9.05,
321
+ 10.42
322
+ ],
323
+ "measured_unix": 1790442895.894016,
324
+ "settings": {
325
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
326
+ "n_gpu_layers": 999,
327
+ "flash_attn": "default (llama.cpp auto on Metal)",
328
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
329
+ },
330
+ "raw": "latency/runs/gguf-q8_0_1024_1q.json"
331
+ },
332
+ {
333
+ "backend": "gguf-q8_0",
334
+ "n_questions": 10,
335
+ "state_tokens": 878,
336
+ "input_tokens_per_question": [
337
+ 943,
338
+ 918,
339
+ 925,
340
+ 911,
341
+ 918,
342
+ 921,
343
+ 943,
344
+ 917,
345
+ 912,
346
+ 919
347
+ ],
348
+ "load_s": 1.0682,
349
+ "cold_s": 1.073,
350
+ "warm_median_s": 1.0243,
351
+ "warm_s": [
352
+ 1.0244,
353
+ 1.0243,
354
+ 1.0231
355
+ ],
356
+ "state_cached_median_s": 0.5268,
357
+ "peak_rss_bytes": {
358
+ "python": 347635712,
359
+ "jev_score_v2": 2757804032
360
+ },
361
+ "peak_phys_footprint_bytes": {
362
+ "python": 258377344,
363
+ "jev_score_v2": 674569792
364
+ },
365
+ "results_identical_cold_vs_state_cached": true,
366
+ "loadavg_at_end": [
367
+ 7.24,
368
+ 8.94,
369
+ 10.37
370
+ ],
371
+ "measured_unix": 1790442903.5177228,
372
+ "settings": {
373
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
374
+ "n_gpu_layers": 999,
375
+ "flash_attn": "default (llama.cpp auto on Metal)",
376
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
377
+ },
378
+ "raw": "latency/runs/gguf-q8_0_1024_10q.json"
379
+ },
380
+ {
381
+ "backend": "gguf-q8_0",
382
+ "n_questions": 1,
383
+ "state_tokens": 3950,
384
+ "input_tokens_per_question": [
385
+ 4015
386
+ ],
387
+ "load_s": 1.0689,
388
+ "cold_s": 2.3987,
389
+ "warm_median_s": 2.3345,
390
+ "warm_s": [
391
+ 2.3333,
392
+ 2.3488,
393
+ 2.3345
394
+ ],
395
+ "state_cached_median_s": 0.082,
396
+ "peak_rss_bytes": {
397
+ "python": 354189312,
398
+ "jev_score_v2": 2807808000
399
+ },
400
+ "peak_phys_footprint_bytes": {
401
+ "python": 247596736,
402
+ "jev_score_v2": 719183680
403
+ },
404
+ "results_identical_cold_vs_state_cached": true,
405
+ "loadavg_at_end": [
406
+ 6.58,
407
+ 8.74,
408
+ 10.28
409
+ ],
410
+ "measured_unix": 1790442915.085064,
411
+ "settings": {
412
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
413
+ "n_gpu_layers": 999,
414
+ "flash_attn": "default (llama.cpp auto on Metal)",
415
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
416
+ },
417
+ "raw": "latency/runs/gguf-q8_0_4096_1q.json"
418
+ },
419
+ {
420
+ "backend": "gguf-q8_0",
421
+ "n_questions": 10,
422
+ "state_tokens": 3950,
423
+ "input_tokens_per_question": [
424
+ 4015,
425
+ 3990,
426
+ 3997,
427
+ 3983,
428
+ 3990,
429
+ 3993,
430
+ 4015,
431
+ 3989,
432
+ 3984,
433
+ 3991
434
+ ],
435
+ "load_s": 1.0684,
436
+ "cold_s": 2.8854,
437
+ "warm_median_s": 2.8178,
438
+ "warm_s": [
439
+ 2.8178,
440
+ 2.8439,
441
+ 2.816
442
+ ],
443
+ "state_cached_median_s": 0.5719,
444
+ "peak_rss_bytes": {
445
+ "python": 319045632,
446
+ "jev_score_v2": 2799091712
447
+ },
448
+ "peak_phys_footprint_bytes": {
449
+ "python": 243910336,
450
+ "jev_score_v2": 716169152
451
+ },
452
+ "results_identical_cold_vs_state_cached": true,
453
+ "loadavg_at_end": [
454
+ 7.36,
455
+ 8.8,
456
+ 10.28
457
+ ],
458
+ "measured_unix": 1790442930.063521,
459
+ "settings": {
460
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
461
+ "n_gpu_layers": 999,
462
+ "flash_attn": "default (llama.cpp auto on Metal)",
463
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
464
+ },
465
+ "raw": "latency/runs/gguf-q8_0_4096_10q.json"
466
+ },
467
+ {
468
+ "backend": "gguf-q8_0",
469
+ "n_questions": 1,
470
+ "state_tokens": 24436,
471
+ "input_tokens_per_question": [
472
+ 24501
473
+ ],
474
+ "load_s": 1.0589,
475
+ "cold_s": 17.0922,
476
+ "warm_median_s": 17.0489,
477
+ "warm_s": [
478
+ 17.0489,
479
+ 17.0376,
480
+ 17.0539
481
+ ],
482
+ "state_cached_median_s": 0.1679,
483
+ "peak_rss_bytes": {
484
+ "python": 364937216,
485
+ "jev_score_v2": 2966470656
486
+ },
487
+ "peak_phys_footprint_bytes": {
488
+ "python": 267110080,
489
+ "jev_score_v2": 875831168
490
+ },
491
+ "results_identical_cold_vs_state_cached": true,
492
+ "loadavg_at_end": [
493
+ 6.7,
494
+ 8.25,
495
+ 9.94
496
+ ],
497
+ "measured_unix": 1790443000.691691,
498
+ "settings": {
499
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
500
+ "n_gpu_layers": 999,
501
+ "flash_attn": "default (llama.cpp auto on Metal)",
502
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
503
+ },
504
+ "raw": "latency/runs/gguf-q8_0_24576_1q.json"
505
+ },
506
+ {
507
+ "backend": "gguf-q8_0",
508
+ "n_questions": 10,
509
+ "state_tokens": 24436,
510
+ "input_tokens_per_question": [
511
+ 24501,
512
+ 24476,
513
+ 24483,
514
+ 24469,
515
+ 24476,
516
+ 24479,
517
+ 24501,
518
+ 24475,
519
+ 24470,
520
+ 24477
521
+ ],
522
+ "load_s": 1.0501,
523
+ "cold_s": 17.8244,
524
+ "warm_median_s": 17.7424,
525
+ "warm_s": [
526
+ 17.7424,
527
+ 17.7312,
528
+ 17.7793
529
+ ],
530
+ "state_cached_median_s": 0.8902,
531
+ "peak_rss_bytes": {
532
+ "python": 348651520,
533
+ "jev_score_v2": 2967109632
534
+ },
535
+ "peak_phys_footprint_bytes": {
536
+ "python": 260425536,
537
+ "jev_score_v2": 878698496
538
+ },
539
+ "results_identical_cold_vs_state_cached": true,
540
+ "loadavg_at_end": [
541
+ 11.28,
542
+ 9.17,
543
+ 10.13
544
+ ],
545
+ "measured_unix": 1790443076.341974,
546
+ "settings": {
547
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
548
+ "n_gpu_layers": 999,
549
+ "flash_attn": "default (llama.cpp auto on Metal)",
550
+ "gguf": "release_2b/gguf/model-q8_0.gguf"
551
+ },
552
+ "raw": "latency/runs/gguf-q8_0_24576_10q.json"
553
+ },
554
+ {
555
+ "backend": "gguf-q4_k_m",
556
+ "n_questions": 1,
557
+ "state_tokens": 878,
558
+ "input_tokens_per_question": [
559
+ 943
560
+ ],
561
+ "load_s": 1.4512,
562
+ "cold_s": 0.6738,
563
+ "warm_median_s": 0.6391,
564
+ "warm_s": [
565
+ 0.6418,
566
+ 0.6391,
567
+ 0.6334
568
+ ],
569
+ "state_cached_median_s": 0.0782,
570
+ "peak_rss_bytes": {
571
+ "python": 334921728,
572
+ "jev_score_v2": 1988542464
573
+ },
574
+ "peak_phys_footprint_bytes": {
575
+ "python": 245073728,
576
+ "jev_score_v2": 669308992
577
+ },
578
+ "results_identical_cold_vs_state_cached": true,
579
+ "loadavg_at_end": [
580
+ 12.77,
581
+ 9.51,
582
+ 10.25
583
+ ],
584
+ "measured_unix": 1790443081.420589,
585
+ "settings": {
586
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
587
+ "n_gpu_layers": 999,
588
+ "flash_attn": "default (llama.cpp auto on Metal)",
589
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
590
+ },
591
+ "raw": "latency/runs/gguf-q4_k_m_1024_1q.json"
592
+ },
593
+ {
594
+ "backend": "gguf-q4_k_m",
595
+ "n_questions": 10,
596
+ "state_tokens": 878,
597
+ "input_tokens_per_question": [
598
+ 943,
599
+ 918,
600
+ 925,
601
+ 911,
602
+ 918,
603
+ 921,
604
+ 943,
605
+ 917,
606
+ 912,
607
+ 919
608
+ ],
609
+ "load_s": 0.9937,
610
+ "cold_s": 1.2045,
611
+ "warm_median_s": 1.157,
612
+ "warm_s": [
613
+ 1.1544,
614
+ 1.1643,
615
+ 1.157
616
+ ],
617
+ "state_cached_median_s": 0.6009,
618
+ "peak_rss_bytes": {
619
+ "python": 352059392,
620
+ "jev_score_v2": 2018754560
621
+ },
622
+ "peak_phys_footprint_bytes": {
623
+ "python": 256673344,
624
+ "jev_score_v2": 665065728
625
+ },
626
+ "results_identical_cold_vs_state_cached": true,
627
+ "loadavg_at_end": [
628
+ 13.83,
629
+ 9.78,
630
+ 10.34
631
+ ],
632
+ "measured_unix": 1790443089.6937559,
633
+ "settings": {
634
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
635
+ "n_gpu_layers": 999,
636
+ "flash_attn": "default (llama.cpp auto on Metal)",
637
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
638
+ },
639
+ "raw": "latency/runs/gguf-q4_k_m_1024_10q.json"
640
+ },
641
+ {
642
+ "backend": "gguf-q4_k_m",
643
+ "n_questions": 1,
644
+ "state_tokens": 3950,
645
+ "input_tokens_per_question": [
646
+ 4015
647
+ ],
648
+ "load_s": 0.9922,
649
+ "cold_s": 2.6608,
650
+ "warm_median_s": 2.6006,
651
+ "warm_s": [
652
+ 2.6006,
653
+ 2.6206,
654
+ 2.5977
655
+ ],
656
+ "state_cached_median_s": 0.0908,
657
+ "peak_rss_bytes": {
658
+ "python": 327221248,
659
+ "jev_score_v2": 2062483456
660
+ },
661
+ "peak_phys_footprint_bytes": {
662
+ "python": 259393408,
663
+ "jev_score_v2": 714692992
664
+ },
665
+ "results_identical_cold_vs_state_cached": true,
666
+ "loadavg_at_end": [
667
+ 11.99,
668
+ 9.57,
669
+ 10.25
670
+ ],
671
+ "measured_unix": 1790443102.1980171,
672
+ "settings": {
673
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
674
+ "n_gpu_layers": 999,
675
+ "flash_attn": "default (llama.cpp auto on Metal)",
676
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
677
+ },
678
+ "raw": "latency/runs/gguf-q4_k_m_4096_1q.json"
679
+ },
680
+ {
681
+ "backend": "gguf-q4_k_m",
682
+ "n_questions": 10,
683
+ "state_tokens": 3950,
684
+ "input_tokens_per_question": [
685
+ 4015,
686
+ 3990,
687
+ 3997,
688
+ 3983,
689
+ 3990,
690
+ 3993,
691
+ 4015,
692
+ 3989,
693
+ 3984,
694
+ 3991
695
+ ],
696
+ "load_s": 0.9859,
697
+ "cold_s": 3.2053,
698
+ "warm_median_s": 3.1651,
699
+ "warm_s": [
700
+ 3.1651,
701
+ 3.1579,
702
+ 3.1711
703
+ ],
704
+ "state_cached_median_s": 0.6462,
705
+ "peak_rss_bytes": {
706
+ "python": 326434816,
707
+ "jev_score_v2": 2073214976
708
+ },
709
+ "peak_phys_footprint_bytes": {
710
+ "python": 245614208,
711
+ "jev_score_v2": 716806400
712
+ },
713
+ "results_identical_cold_vs_state_cached": true,
714
+ "loadavg_at_end": [
715
+ 10.28,
716
+ 9.31,
717
+ 10.15
718
+ ],
719
+ "measured_unix": 1790443118.634955,
720
+ "settings": {
721
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
722
+ "n_gpu_layers": 999,
723
+ "flash_attn": "default (llama.cpp auto on Metal)",
724
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
725
+ },
726
+ "raw": "latency/runs/gguf-q4_k_m_4096_10q.json"
727
+ },
728
+ {
729
+ "backend": "gguf-q4_k_m",
730
+ "n_questions": 1,
731
+ "state_tokens": 24436,
732
+ "input_tokens_per_question": [
733
+ 24501
734
+ ],
735
+ "load_s": 1.0041,
736
+ "cold_s": 18.6914,
737
+ "warm_median_s": 18.6165,
738
+ "warm_s": [
739
+ 18.6165,
740
+ 18.6059,
741
+ 18.624
742
+ ],
743
+ "state_cached_median_s": 0.1755,
744
+ "peak_rss_bytes": {
745
+ "python": 345686016,
746
+ "jev_score_v2": 2233663488
747
+ },
748
+ "peak_phys_footprint_bytes": {
749
+ "python": 260048576,
750
+ "jev_score_v2": 880335552
751
+ },
752
+ "results_identical_cold_vs_state_cached": true,
753
+ "loadavg_at_end": [
754
+ 8.43,
755
+ 9.22,
756
+ 10.05
757
+ ],
758
+ "measured_unix": 1790443195.534742,
759
+ "settings": {
760
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
761
+ "n_gpu_layers": 999,
762
+ "flash_attn": "default (llama.cpp auto on Metal)",
763
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
764
+ },
765
+ "raw": "latency/runs/gguf-q4_k_m_24576_1q.json"
766
+ },
767
+ {
768
+ "backend": "gguf-q4_k_m",
769
+ "n_questions": 10,
770
+ "state_tokens": 24436,
771
+ "input_tokens_per_question": [
772
+ 24501,
773
+ 24476,
774
+ 24483,
775
+ 24469,
776
+ 24476,
777
+ 24479,
778
+ 24501,
779
+ 24475,
780
+ 24470,
781
+ 24477
782
+ ],
783
+ "load_s": 0.9967,
784
+ "cold_s": 19.4776,
785
+ "warm_median_s": 19.4309,
786
+ "warm_s": [
787
+ 19.4309,
788
+ 19.4251,
789
+ 19.5057
790
+ ],
791
+ "state_cached_median_s": 0.9616,
792
+ "peak_rss_bytes": {
793
+ "python": 354992128,
794
+ "jev_score_v2": 2226913280
795
+ },
796
+ "peak_phys_footprint_bytes": {
797
+ "python": 250791552,
798
+ "jev_score_v2": 870128320
799
+ },
800
+ "results_identical_cold_vs_state_cached": true,
801
+ "loadavg_at_end": [
802
+ 5.63,
803
+ 8.19,
804
+ 9.59
805
+ ],
806
+ "measured_unix": 1790443278.075001,
807
+ "settings": {
808
+ "engine": "jev-score-v2 (staged jev_score_v2.cpp, llama.cpp 441df11f, Metal)",
809
+ "n_gpu_layers": 999,
810
+ "flash_attn": "default (llama.cpp auto on Metal)",
811
+ "gguf": "release_2b/gguf/model-q4_k_m.gguf"
812
+ },
813
+ "raw": "latency/runs/gguf-q4_k_m_24576_10q.json"
814
+ },
815
+ {
816
+ "backend": "mlx-bf16",
817
+ "n_questions": 1,
818
+ "state_tokens": 878,
819
+ "input_tokens_per_question": [
820
+ 943
821
+ ],
822
+ "load_s": 3.6602,
823
+ "cold_s": 0.64,
824
+ "warm_median_s": 0.5658,
825
+ "warm_s": [
826
+ 0.5615,
827
+ 0.5665,
828
+ 0.5658
829
+ ],
830
+ "state_cached_median_s": 0.0796,
831
+ "peak_rss_bytes": {
832
+ "python": 4401463296
833
+ },
834
+ "peak_phys_footprint_bytes": {
835
+ "python": 5205486080
836
+ },
837
+ "mlx_peak_memory_bytes": 4536436474,
838
+ "results_identical_cold_vs_state_cached": true,
839
+ "loadavg_at_end": [
840
+ 6.5,
841
+ 8.17,
842
+ 9.53
843
+ ],
844
+ "measured_unix": 1790443305.918901,
845
+ "settings": {
846
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
847
+ "precision": "bf16",
848
+ "compute_dtype": null
849
+ },
850
+ "raw": "latency/runs/mlx-bf16_1024_1q.json"
851
+ },
852
+ {
853
+ "backend": "mlx-bf16",
854
+ "n_questions": 10,
855
+ "state_tokens": 878,
856
+ "input_tokens_per_question": [
857
+ 943,
858
+ 918,
859
+ 925,
860
+ 911,
861
+ 918,
862
+ 921,
863
+ 943,
864
+ 917,
865
+ 912,
866
+ 919
867
+ ],
868
+ "load_s": 3.3114,
869
+ "cold_s": 1.1022,
870
+ "warm_median_s": 1.0651,
871
+ "warm_s": [
872
+ 1.0651,
873
+ 1.0647,
874
+ 1.0714
875
+ ],
876
+ "state_cached_median_s": 0.5726,
877
+ "peak_rss_bytes": {
878
+ "python": 4416684032
879
+ },
880
+ "peak_phys_footprint_bytes": {
881
+ "python": 5575518912
882
+ },
883
+ "mlx_peak_memory_bytes": 4536436474,
884
+ "results_identical_cold_vs_state_cached": true,
885
+ "loadavg_at_end": [
886
+ 6.03,
887
+ 8.01,
888
+ 9.46
889
+ ],
890
+ "measured_unix": 1790443316.3707972,
891
+ "settings": {
892
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
893
+ "precision": "bf16",
894
+ "compute_dtype": null
895
+ },
896
+ "raw": "latency/runs/mlx-bf16_1024_10q.json"
897
+ },
898
+ {
899
+ "backend": "mlx-bf16",
900
+ "n_questions": 1,
901
+ "state_tokens": 3950,
902
+ "input_tokens_per_question": [
903
+ 4015
904
+ ],
905
+ "load_s": 3.2848,
906
+ "cold_s": 2.265,
907
+ "warm_median_s": 2.2579,
908
+ "warm_s": [
909
+ 2.2462,
910
+ 2.2579,
911
+ 2.2701
912
+ ],
913
+ "state_cached_median_s": 0.0903,
914
+ "peak_rss_bytes": {
915
+ "python": 4408279040
916
+ },
917
+ "peak_phys_footprint_bytes": {
918
+ "python": 6899099712
919
+ },
920
+ "mlx_peak_memory_bytes": 5062198966,
921
+ "results_identical_cold_vs_state_cached": true,
922
+ "loadavg_at_end": [
923
+ 5.78,
924
+ 7.9,
925
+ 9.4
926
+ ],
927
+ "measured_unix": 1790443330.072621,
928
+ "settings": {
929
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
930
+ "precision": "bf16",
931
+ "compute_dtype": null
932
+ },
933
+ "raw": "latency/runs/mlx-bf16_4096_1q.json"
934
+ },
935
+ {
936
+ "backend": "mlx-bf16",
937
+ "n_questions": 10,
938
+ "state_tokens": 3950,
939
+ "input_tokens_per_question": [
940
+ 4015,
941
+ 3990,
942
+ 3997,
943
+ 3983,
944
+ 3990,
945
+ 3993,
946
+ 4015,
947
+ 3989,
948
+ 3984,
949
+ 3991
950
+ ],
951
+ "load_s": 3.3381,
952
+ "cold_s": 2.8286,
953
+ "warm_median_s": 2.7998,
954
+ "warm_s": [
955
+ 2.832,
956
+ 2.7968,
957
+ 2.7998
958
+ ],
959
+ "state_cached_median_s": 0.622,
960
+ "peak_rss_bytes": {
961
+ "python": 4406149120
962
+ },
963
+ "peak_phys_footprint_bytes": {
964
+ "python": 6943402624
965
+ },
966
+ "mlx_peak_memory_bytes": 5062395574,
967
+ "results_identical_cold_vs_state_cached": true,
968
+ "loadavg_at_end": [
969
+ 5.53,
970
+ 7.71,
971
+ 9.3
972
+ ],
973
+ "measured_unix": 1790443347.640775,
974
+ "settings": {
975
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
976
+ "precision": "bf16",
977
+ "compute_dtype": null
978
+ },
979
+ "raw": "latency/runs/mlx-bf16_4096_10q.json"
980
+ },
981
+ {
982
+ "backend": "mlx-bf16",
983
+ "n_questions": 1,
984
+ "state_tokens": 24436,
985
+ "input_tokens_per_question": [
986
+ 24501
987
+ ],
988
+ "load_s": 3.2657,
989
+ "cold_s": 15.472,
990
+ "warm_median_s": 15.5073,
991
+ "warm_s": [
992
+ 15.5202,
993
+ 15.5073,
994
+ 15.4734
995
+ ],
996
+ "state_cached_median_s": 0.1546,
997
+ "peak_rss_bytes": {
998
+ "python": 4403707904
999
+ },
1000
+ "peak_phys_footprint_bytes": {
1001
+ "python": 7889725824
1002
+ },
1003
+ "mlx_peak_memory_bytes": 5942822582,
1004
+ "results_identical_cold_vs_state_cached": true,
1005
+ "loadavg_at_end": [
1006
+ 6.15,
1007
+ 7.49,
1008
+ 9.1
1009
+ ],
1010
+ "measured_unix": 1790443414.47198,
1011
+ "settings": {
1012
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1013
+ "precision": "bf16",
1014
+ "compute_dtype": null
1015
+ },
1016
+ "raw": "latency/runs/mlx-bf16_24576_1q.json"
1017
+ },
1018
+ {
1019
+ "backend": "mlx-bf16",
1020
+ "n_questions": 10,
1021
+ "state_tokens": 24436,
1022
+ "input_tokens_per_question": [
1023
+ 24501,
1024
+ 24476,
1025
+ 24483,
1026
+ 24469,
1027
+ 24476,
1028
+ 24479,
1029
+ 24501,
1030
+ 24475,
1031
+ 24470,
1032
+ 24477
1033
+ ],
1034
+ "load_s": 3.278,
1035
+ "cold_s": 16.2813,
1036
+ "warm_median_s": 16.2351,
1037
+ "warm_s": [
1038
+ 16.2243,
1039
+ 16.2351,
1040
+ 16.2657
1041
+ ],
1042
+ "state_cached_median_s": 0.8797,
1043
+ "peak_rss_bytes": {
1044
+ "python": 4405805056
1045
+ },
1046
+ "peak_phys_footprint_bytes": {
1047
+ "python": 7863035904
1048
+ },
1049
+ "mlx_peak_memory_bytes": 5942806198,
1050
+ "results_identical_cold_vs_state_cached": true,
1051
+ "loadavg_at_end": [
1052
+ 8.07,
1053
+ 7.75,
1054
+ 9.06
1055
+ ],
1056
+ "measured_unix": 1790443486.522902,
1057
+ "settings": {
1058
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1059
+ "precision": "bf16",
1060
+ "compute_dtype": null
1061
+ },
1062
+ "raw": "latency/runs/mlx-bf16_24576_10q.json"
1063
+ },
1064
+ {
1065
+ "backend": "mlx-8bit",
1066
+ "n_questions": 1,
1067
+ "state_tokens": 878,
1068
+ "input_tokens_per_question": [
1069
+ 943
1070
+ ],
1071
+ "load_s": 3.3345,
1072
+ "cold_s": 0.7573,
1073
+ "warm_median_s": 0.72,
1074
+ "warm_s": [
1075
+ 0.7177,
1076
+ 0.72,
1077
+ 0.7228
1078
+ ],
1079
+ "state_cached_median_s": 0.0834,
1080
+ "peak_rss_bytes": {
1081
+ "python": 2640986112
1082
+ },
1083
+ "peak_phys_footprint_bytes": {
1084
+ "python": 3772211392
1085
+ },
1086
+ "mlx_peak_memory_bytes": 3074287404,
1087
+ "results_identical_cold_vs_state_cached": true,
1088
+ "loadavg_at_end": [
1089
+ 8.14,
1090
+ 7.77,
1091
+ 9.06
1092
+ ],
1093
+ "measured_unix": 1790443494.139926,
1094
+ "settings": {
1095
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1096
+ "precision": "8bit",
1097
+ "compute_dtype": null
1098
+ },
1099
+ "raw": "latency/runs/mlx-8bit_1024_1q.json"
1100
+ },
1101
+ {
1102
+ "backend": "mlx-8bit",
1103
+ "n_questions": 10,
1104
+ "state_tokens": 878,
1105
+ "input_tokens_per_question": [
1106
+ 943,
1107
+ 918,
1108
+ 925,
1109
+ 911,
1110
+ 918,
1111
+ 921,
1112
+ 943,
1113
+ 917,
1114
+ 912,
1115
+ 919
1116
+ ],
1117
+ "load_s": 3.1502,
1118
+ "cold_s": 1.3124,
1119
+ "warm_median_s": 1.2657,
1120
+ "warm_s": [
1121
+ 1.2719,
1122
+ 1.2657,
1123
+ 1.2621
1124
+ ],
1125
+ "state_cached_median_s": 0.6288,
1126
+ "peak_rss_bytes": {
1127
+ "python": 2635268096
1128
+ },
1129
+ "peak_phys_footprint_bytes": {
1130
+ "python": 4171391040
1131
+ },
1132
+ "mlx_peak_memory_bytes": 3074287404,
1133
+ "results_identical_cold_vs_state_cached": true,
1134
+ "loadavg_at_end": [
1135
+ 7.51,
1136
+ 7.65,
1137
+ 9.0
1138
+ ],
1139
+ "measured_unix": 1790443505.379929,
1140
+ "settings": {
1141
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1142
+ "precision": "8bit",
1143
+ "compute_dtype": null
1144
+ },
1145
+ "raw": "latency/runs/mlx-8bit_1024_10q.json"
1146
+ },
1147
+ {
1148
+ "backend": "mlx-8bit",
1149
+ "n_questions": 1,
1150
+ "state_tokens": 3950,
1151
+ "input_tokens_per_question": [
1152
+ 4015
1153
+ ],
1154
+ "load_s": 3.1569,
1155
+ "cold_s": 2.9904,
1156
+ "warm_median_s": 2.9581,
1157
+ "warm_s": [
1158
+ 2.9525,
1159
+ 2.9624,
1160
+ 2.9581
1161
+ ],
1162
+ "state_cached_median_s": 0.0922,
1163
+ "peak_rss_bytes": {
1164
+ "python": 2639839232
1165
+ },
1166
+ "peak_phys_footprint_bytes": {
1167
+ "python": 5536832896
1168
+ },
1169
+ "mlx_peak_memory_bytes": 3471550184,
1170
+ "results_identical_cold_vs_state_cached": true,
1171
+ "loadavg_at_end": [
1172
+ 8.57,
1173
+ 7.86,
1174
+ 9.04
1175
+ ],
1176
+ "measured_unix": 1790443521.78049,
1177
+ "settings": {
1178
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1179
+ "precision": "8bit",
1180
+ "compute_dtype": null
1181
+ },
1182
+ "raw": "latency/runs/mlx-8bit_4096_1q.json"
1183
+ },
1184
+ {
1185
+ "backend": "mlx-8bit",
1186
+ "n_questions": 10,
1187
+ "state_tokens": 3950,
1188
+ "input_tokens_per_question": [
1189
+ 4015,
1190
+ 3990,
1191
+ 3997,
1192
+ 3983,
1193
+ 3990,
1194
+ 3993,
1195
+ 4015,
1196
+ 3989,
1197
+ 3984,
1198
+ 3991
1199
+ ],
1200
+ "load_s": 3.1665,
1201
+ "cold_s": 3.572,
1202
+ "warm_median_s": 3.5432,
1203
+ "warm_s": [
1204
+ 3.5584,
1205
+ 3.5432,
1206
+ 3.5354
1207
+ ],
1208
+ "state_cached_median_s": 0.6768,
1209
+ "peak_rss_bytes": {
1210
+ "python": 2624536576
1211
+ },
1212
+ "peak_phys_footprint_bytes": {
1213
+ "python": 5698149760
1214
+ },
1215
+ "mlx_peak_memory_bytes": 3471615720,
1216
+ "results_identical_cold_vs_state_cached": true,
1217
+ "loadavg_at_end": [
1218
+ 7.82,
1219
+ 7.73,
1220
+ 8.97
1221
+ ],
1222
+ "measured_unix": 1790443542.268524,
1223
+ "settings": {
1224
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1225
+ "precision": "8bit",
1226
+ "compute_dtype": null
1227
+ },
1228
+ "raw": "latency/runs/mlx-8bit_4096_10q.json"
1229
+ },
1230
+ {
1231
+ "backend": "mlx-8bit",
1232
+ "n_questions": 1,
1233
+ "state_tokens": 24436,
1234
+ "input_tokens_per_question": [
1235
+ 24501
1236
+ ],
1237
+ "load_s": 3.1447,
1238
+ "cold_s": 19.75,
1239
+ "warm_median_s": 19.8614,
1240
+ "warm_s": [
1241
+ 19.8005,
1242
+ 19.8614,
1243
+ 19.9115
1244
+ ],
1245
+ "state_cached_median_s": 0.1562,
1246
+ "peak_rss_bytes": {
1247
+ "python": 2643197952
1248
+ },
1249
+ "peak_phys_footprint_bytes": {
1250
+ "python": 6518807424
1251
+ },
1252
+ "mlx_peak_memory_bytes": 4350994102,
1253
+ "results_identical_cold_vs_state_cached": true,
1254
+ "loadavg_at_end": [
1255
+ 6.7,
1256
+ 7.34,
1257
+ 8.69
1258
+ ],
1259
+ "measured_unix": 1790443626.3357399,
1260
+ "settings": {
1261
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1262
+ "precision": "8bit",
1263
+ "compute_dtype": null
1264
+ },
1265
+ "raw": "latency/runs/mlx-8bit_24576_1q.json"
1266
+ },
1267
+ {
1268
+ "backend": "mlx-8bit",
1269
+ "n_questions": 10,
1270
+ "state_tokens": 24436,
1271
+ "input_tokens_per_question": [
1272
+ 24501,
1273
+ 24476,
1274
+ 24483,
1275
+ 24469,
1276
+ 24476,
1277
+ 24479,
1278
+ 24501,
1279
+ 24475,
1280
+ 24470,
1281
+ 24477
1282
+ ],
1283
+ "load_s": 3.1668,
1284
+ "cold_s": 20.7629,
1285
+ "warm_median_s": 20.7793,
1286
+ "warm_s": [
1287
+ 20.6782,
1288
+ 20.7793,
1289
+ 21.8013
1290
+ ],
1291
+ "state_cached_median_s": 0.9543,
1292
+ "peak_rss_bytes": {
1293
+ "python": 2643705856
1294
+ },
1295
+ "peak_phys_footprint_bytes": {
1296
+ "python": 6522985664
1297
+ },
1298
+ "mlx_peak_memory_bytes": 4350944950,
1299
+ "results_identical_cold_vs_state_cached": true,
1300
+ "loadavg_at_end": [
1301
+ 8.26,
1302
+ 8.35,
1303
+ 8.99
1304
+ ],
1305
+ "measured_unix": 1790443717.49426,
1306
+ "settings": {
1307
+ "engine": "mlx 0.32.2 / mlx-lm 0.31.3 (staged runtime)",
1308
+ "precision": "8bit",
1309
+ "compute_dtype": null
1310
+ },
1311
+ "raw": "latency/runs/mlx-8bit_24576_10q.json"
1312
+ },
1313
+ {
1314
+ "backend": "torch-mps-fp32",
1315
+ "n_questions": 1,
1316
+ "state_tokens": 878,
1317
+ "input_tokens_per_question": [
1318
+ 943
1319
+ ],
1320
+ "load_s": 7.2396,
1321
+ "cold_s": 2.0776,
1322
+ "warm_median_s": 1.82,
1323
+ "warm_s": [
1324
+ 1.82,
1325
+ 1.8185,
1326
+ 1.8657
1327
+ ],
1328
+ "state_cached_median_s": 0.2243,
1329
+ "peak_rss_bytes": {
1330
+ "python": 11777310720
1331
+ },
1332
+ "peak_phys_footprint_bytes": {
1333
+ "python": 9488876864
1334
+ },
1335
+ "results_identical_cold_vs_state_cached": true,
1336
+ "loadavg_at_end": [
1337
+ 11.08,
1338
+ 15.97,
1339
+ 14.21
1340
+ ],
1341
+ "measured_unix": 1790433935.3785548,
1342
+ "settings": {
1343
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1344
+ "device": "mps",
1345
+ "dtype": "float32",
1346
+ "attn_chunk": 1024
1347
+ },
1348
+ "raw": "latency/runs/torch-mps-fp32_1024_1q.json"
1349
+ },
1350
+ {
1351
+ "backend": "torch-mps-fp32",
1352
+ "n_questions": 10,
1353
+ "state_tokens": 878,
1354
+ "input_tokens_per_question": [
1355
+ 943,
1356
+ 918,
1357
+ 925,
1358
+ 911,
1359
+ 918,
1360
+ 921,
1361
+ 943,
1362
+ 917,
1363
+ 912,
1364
+ 919
1365
+ ],
1366
+ "load_s": 5.8844,
1367
+ "cold_s": 3.487,
1368
+ "warm_median_s": 3.126,
1369
+ "warm_s": [
1370
+ 3.126,
1371
+ 3.124,
1372
+ 3.1357
1373
+ ],
1374
+ "state_cached_median_s": 1.534,
1375
+ "peak_rss_bytes": {
1376
+ "python": 11766398976
1377
+ },
1378
+ "peak_phys_footprint_bytes": {
1379
+ "python": 9348826496
1380
+ },
1381
+ "results_identical_cold_vs_state_cached": true,
1382
+ "loadavg_at_end": [
1383
+ 10.92,
1384
+ 15.56,
1385
+ 14.11
1386
+ ],
1387
+ "measured_unix": 1790433960.3376808,
1388
+ "settings": {
1389
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1390
+ "device": "mps",
1391
+ "dtype": "float32",
1392
+ "attn_chunk": 1024
1393
+ },
1394
+ "raw": "latency/runs/torch-mps-fp32_1024_10q.json"
1395
+ },
1396
+ {
1397
+ "backend": "torch-mps-fp32",
1398
+ "n_questions": 1,
1399
+ "state_tokens": 3950,
1400
+ "input_tokens_per_question": [
1401
+ 4015
1402
+ ],
1403
+ "load_s": 6.6133,
1404
+ "cold_s": 7.9626,
1405
+ "warm_median_s": 7.7385,
1406
+ "warm_s": [
1407
+ 7.7135,
1408
+ 7.7385,
1409
+ 7.7569
1410
+ ],
1411
+ "state_cached_median_s": 0.2508,
1412
+ "peak_rss_bytes": {
1413
+ "python": 11774099456
1414
+ },
1415
+ "peak_phys_footprint_bytes": {
1416
+ "python": 9830892736
1417
+ },
1418
+ "results_identical_cold_vs_state_cached": true,
1419
+ "loadavg_at_end": [
1420
+ 8.41,
1421
+ 7.52,
1422
+ 7.06
1423
+ ],
1424
+ "measured_unix": 1790438685.325036,
1425
+ "settings": {
1426
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1427
+ "device": "mps",
1428
+ "dtype": "float32",
1429
+ "attn_chunk": 1024
1430
+ },
1431
+ "raw": "latency/runs/torch-mps-fp32_4096_1q.json"
1432
+ },
1433
+ {
1434
+ "backend": "torch-mps-fp32",
1435
+ "n_questions": 10,
1436
+ "state_tokens": 3950,
1437
+ "input_tokens_per_question": [
1438
+ 4015,
1439
+ 3990,
1440
+ 3997,
1441
+ 3983,
1442
+ 3990,
1443
+ 3993,
1444
+ 4015,
1445
+ 3989,
1446
+ 3984,
1447
+ 3991
1448
+ ],
1449
+ "load_s": 5.8354,
1450
+ "cold_s": 9.4943,
1451
+ "warm_median_s": 9.2321,
1452
+ "warm_s": [
1453
+ 9.1731,
1454
+ 9.2321,
1455
+ 9.2805
1456
+ ],
1457
+ "state_cached_median_s": 1.7373,
1458
+ "peak_rss_bytes": {
1459
+ "python": 11766579200
1460
+ },
1461
+ "peak_phys_footprint_bytes": {
1462
+ "python": 9740780928
1463
+ },
1464
+ "results_identical_cold_vs_state_cached": true,
1465
+ "loadavg_at_end": [
1466
+ 10.4,
1467
+ 8.12,
1468
+ 7.31
1469
+ ],
1470
+ "measured_unix": 1790438734.979529,
1471
+ "settings": {
1472
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1473
+ "device": "mps",
1474
+ "dtype": "float32",
1475
+ "attn_chunk": 1024
1476
+ },
1477
+ "raw": "latency/runs/torch-mps-fp32_4096_10q.json"
1478
+ },
1479
+ {
1480
+ "backend": "torch-mps-fp32",
1481
+ "n_questions": 1,
1482
+ "state_tokens": 24436,
1483
+ "input_tokens_per_question": [
1484
+ 24501
1485
+ ],
1486
+ "load_s": 7.5475,
1487
+ "cold_s": 56.4354,
1488
+ "warm_median_s": 61.9146,
1489
+ "warm_s": [
1490
+ 55.156,
1491
+ 61.9146,
1492
+ 80.3434
1493
+ ],
1494
+ "state_cached_median_s": 0.6,
1495
+ "peak_rss_bytes": {
1496
+ "python": 11774935040
1497
+ },
1498
+ "peak_phys_footprint_bytes": {
1499
+ "python": 13703111296
1500
+ },
1501
+ "results_identical_cold_vs_state_cached": true,
1502
+ "loadavg_at_end": [
1503
+ 7.27,
1504
+ 7.84,
1505
+ 7.44
1506
+ ],
1507
+ "measured_unix": 1790438999.6904268,
1508
+ "settings": {
1509
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1510
+ "device": "mps",
1511
+ "dtype": "float32",
1512
+ "attn_chunk": 1024
1513
+ },
1514
+ "raw": "latency/runs/torch-mps-fp32_24576_1q.json"
1515
+ },
1516
+ {
1517
+ "backend": "torch-mps-fp32",
1518
+ "n_questions": 10,
1519
+ "state_tokens": 24436,
1520
+ "input_tokens_per_question": [
1521
+ 24501,
1522
+ 24476,
1523
+ 24483,
1524
+ 24469,
1525
+ 24476,
1526
+ 24479,
1527
+ 24501,
1528
+ 24475,
1529
+ 24470,
1530
+ 24477
1531
+ ],
1532
+ "load_s": 7.9699,
1533
+ "cold_s": 72.6584,
1534
+ "warm_median_s": 72.3035,
1535
+ "warm_s": [
1536
+ 83.2192,
1537
+ 72.3035,
1538
+ 58.3757
1539
+ ],
1540
+ "state_cached_median_s": 2.7217,
1541
+ "peak_rss_bytes": {
1542
+ "python": 11775229952
1543
+ },
1544
+ "peak_phys_footprint_bytes": {
1545
+ "python": 13640000128
1546
+ },
1547
+ "results_identical_cold_vs_state_cached": true,
1548
+ "loadavg_at_end": [
1549
+ 9.6,
1550
+ 8.35,
1551
+ 7.72
1552
+ ],
1553
+ "measured_unix": 1790439304.235886,
1554
+ "settings": {
1555
+ "engine": "transformers 5.17 / torch 2.14 (staged runtime)",
1556
+ "device": "mps",
1557
+ "dtype": "float32",
1558
+ "attn_chunk": 1024
1559
+ },
1560
+ "raw": "latency/runs/torch-mps-fp32_24576_10q.json"
1561
+ }
1562
+ ],
1563
+ "scripts": [
1564
+ "release_2b/latency/latency_run.py",
1565
+ "release_2b/latency/run_all.sh",
1566
+ "release_2b/latency/aggregate.py"
1567
+ ]
1568
+ }
validation/parity/PREDECLARED_RELEASE_GATES_2B.md ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Jev-Style-2B-Decision-v3 — release format gates (pre-declared 2026-09-26, BEFORE any trained-weight format was scored)
2
+
3
+ Reference: HF FP32 CPU, exact v2 block attention, on the released bf16 text-only checkpoint
4
+ runs/macjev/candidate_2b/hf-candidate (SHA256SUMS 279/279 verified). Temperature T = 0.8278650621 (macjev_readout.json).
5
+
6
+ Fixtures (frozen, sha256 in their .stats/.README):
7
+ - base fixture: runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl (35 req / 43 q, 214,700 tok, up to 25,600 tok, overflow, K<=151)
8
+ - gate fixture: runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl (1,000 real dev rows, <=4,096 tok)
9
+
10
+ Gates (training plan §11.2, unchanged): F16/bf16 top-1 >= .99; 8-bit top-1 >= .98; |dNLL| <= .02 (fp/16/8-bit);
11
+ accuracy drop Q8_0/MLX-8bit <= 0.3 pp, Q4_K_M <= 1.0 pp (measured on the gate fixture; n=1000 => 0.1 pp per flip).
12
+ Block-mask proof on the base fixture: every question closer to the block reference than to the causal and prefix-causal controls.
13
+
14
+ Release set (user choice): GGUF F16 / Q8_0 / Q4_K_M; MLX bf16 / 8bit (affine g64).
15
+ Recipe 1 (primary): stock llama-quantize defaults; mlx_lm.convert defaults + FP32 norm sidecar + MLXScorerV2.
16
+ Recipe 2 (declared fallback, used ONLY if recipe 1 fails a gate): keep the tied embedding / readout rows at f16
17
+ (llama-quantize --token-embedding-type f16; MLX: quant predicate excluding embed_tokens). Same gates, same fixtures.
18
+ If recipe 2 also fails, that format is NOT shipped. Gates are not relaxed after seeing results.
19
+
20
+ ## Public benchmarks (pre-declared 2026-09-26 22:01:49 AEST, before any 2B benchmark item was scored)
21
+ Timing (erratum added 2026-09-27 03:35 AEST): this section was appended by the same shell command that launched the
22
+ benchmark runs, immediately before the first item (file mtime 22:01:49; run log `runs/macjev/bench_2b/run_trained.log`
23
+ first line `[2026-09-26 22:01:49] START jevbench`; first result seen 22:06:28). The heading originally said
24
+ "22:10" by mistake. The format-gate section above was written when the file was created (21:46:03), before any
25
+ trained-weight format was scored (first format score started 21:53:00).
26
+ - Reported engine: GGUF F16 (runs/macjev/release_2b/gguf/model-f16.gguf, jev-score-v2, Metal), global T = 0.8278650621 only.
27
+ - JevBench v1.4.1 public (231) and elcronos zero-shot sets (tweet_topic / fin_topic / daily_dialog), each run ONCE,
28
+ same metric code as the 0.8B v3 card. No per-benchmark temperatures, no reruns, no prompt changes after seeing scores.
29
+ - Decision Index 0.2: not run locally; requested from the maintainer after release (user decision 2026-09-26).
validation/parity/compare_mlx_affine8-g64_base_vs_block.json ADDED
@@ -0,0 +1,276 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "8bit",
3
+ "temperature": 0.8278650620942867,
4
+ "overall": {
5
+ "n": 43,
6
+ "top1_agreement": 1.0,
7
+ "top1_flips": 0,
8
+ "max_abs_dscore": 0.4653351306915283,
9
+ "mean_abs_dscore": 0.023077200647273945,
10
+ "mean_kl": 0.0006975644872166873,
11
+ "n_gold": 43,
12
+ "nll_ref": 0.670433307194272,
13
+ "nll_cand": 0.6868139442506376,
14
+ "dnll": 0.01638063705636561,
15
+ "mean_abs_row_dnll": 0.030139399526949388,
16
+ "acc_ref": 0.7209302325581395,
17
+ "acc_cand": 0.7209302325581395,
18
+ "acc_drop_pp": 0.0
19
+ },
20
+ "per_category": {
21
+ "catalogue_overflow": {
22
+ "n": 3,
23
+ "top1_agreement": 1.0,
24
+ "top1_flips": 0,
25
+ "max_abs_dscore": 0.2735496163368225,
26
+ "mean_abs_dscore": 0.019241230377298316,
27
+ "mean_kl": 0.0012829640019479873,
28
+ "n_gold": 3,
29
+ "nll_ref": 0.5023960827418966,
30
+ "nll_cand": 0.46349011750257935,
31
+ "dnll": -0.038905965239317275,
32
+ "mean_abs_row_dnll": 0.039379824994875905,
33
+ "acc_ref": 0.6666666666666666,
34
+ "acc_cand": 0.6666666666666666,
35
+ "acc_drop_pp": 0.0
36
+ },
37
+ "catalogue_overflow_long_state": {
38
+ "n": 1,
39
+ "top1_agreement": 1.0,
40
+ "top1_flips": 0,
41
+ "max_abs_dscore": 0.09829854965209961,
42
+ "mean_abs_dscore": 0.02608197888001701,
43
+ "mean_kl": 0.00037418306666026774,
44
+ "n_gold": 1,
45
+ "nll_ref": 1.868953922917252,
46
+ "nll_cand": 1.8305656122814864,
47
+ "dnll": -0.03838831063576564,
48
+ "mean_abs_row_dnll": 0.03838831063576564,
49
+ "acc_ref": 0.0,
50
+ "acc_cand": 0.0,
51
+ "acc_drop_pp": 0.0
52
+ },
53
+ "long_16k": {
54
+ "n": 3,
55
+ "top1_agreement": 1.0,
56
+ "top1_flips": 0,
57
+ "max_abs_dscore": 0.4653351306915283,
58
+ "mean_abs_dscore": 0.04898447410160343,
59
+ "mean_kl": 0.005652873428140343,
60
+ "n_gold": 3,
61
+ "nll_ref": 1.098996986024778,
62
+ "nll_cand": 1.3040987437659517,
63
+ "dnll": 0.20510175774117378,
64
+ "mean_abs_row_dnll": 0.2051017577411738,
65
+ "acc_ref": 0.6666666666666666,
66
+ "acc_cand": 0.6666666666666666,
67
+ "acc_drop_pp": 0.0
68
+ },
69
+ "long_24k": {
70
+ "n": 2,
71
+ "top1_agreement": 1.0,
72
+ "top1_flips": 0,
73
+ "max_abs_dscore": 0.068747878074646,
74
+ "mean_abs_dscore": 0.03896120687325796,
75
+ "mean_kl": 0.0007690578648421433,
76
+ "n_gold": 2,
77
+ "nll_ref": 0.43411292245601785,
78
+ "nll_cand": 0.43100727732191557,
79
+ "dnll": -0.003105645134102275,
80
+ "mean_abs_row_dnll": 0.019953016950989222,
81
+ "acc_ref": 1.0,
82
+ "acc_cand": 1.0,
83
+ "acc_drop_pp": 0.0
84
+ },
85
+ "many_options": {
86
+ "n": 2,
87
+ "top1_agreement": 1.0,
88
+ "top1_flips": 0,
89
+ "max_abs_dscore": 0.06005406379699707,
90
+ "mean_abs_dscore": 0.01609148249030113,
91
+ "mean_kl": 3.102768330017338e-06,
92
+ "n_gold": 2,
93
+ "nll_ref": 0.012124599411053285,
94
+ "nll_cand": 0.012000711496263293,
95
+ "dnll": -0.00012388791478999163,
96
+ "mean_abs_row_dnll": 0.00012388791478999076,
97
+ "acc_ref": 1.0,
98
+ "acc_cand": 1.0,
99
+ "acc_drop_pp": 0.0
100
+ },
101
+ "multi_block_3k": {
102
+ "n": 2,
103
+ "top1_agreement": 1.0,
104
+ "top1_flips": 0,
105
+ "max_abs_dscore": 0.037885069847106934,
106
+ "mean_abs_dscore": 0.015739992260932922,
107
+ "mean_kl": 0.00011035902286322221,
108
+ "n_gold": 2,
109
+ "nll_ref": 0.10264503770560551,
110
+ "nll_cand": 0.09882922575086513,
111
+ "dnll": -0.0038158119547403863,
112
+ "mean_abs_row_dnll": 0.00429795147361977,
113
+ "acc_ref": 1.0,
114
+ "acc_cand": 1.0,
115
+ "acc_drop_pp": 0.0
116
+ },
117
+ "multi_block_5k": {
118
+ "n": 2,
119
+ "top1_agreement": 1.0,
120
+ "top1_flips": 0,
121
+ "max_abs_dscore": 0.0428202748298645,
122
+ "mean_abs_dscore": 0.02182019054889679,
123
+ "mean_kl": 4.99818269133305e-05,
124
+ "n_gold": 2,
125
+ "nll_ref": 0.41414880456154163,
126
+ "nll_cand": 0.40767085022183286,
127
+ "dnll": -0.006477954339708769,
128
+ "mean_abs_row_dnll": 0.006477954339708783,
129
+ "acc_ref": 1.0,
130
+ "acc_cand": 1.0,
131
+ "acc_drop_pp": 0.0
132
+ },
133
+ "multi_block_9k": {
134
+ "n": 2,
135
+ "top1_agreement": 1.0,
136
+ "top1_flips": 0,
137
+ "max_abs_dscore": 0.02951335906982422,
138
+ "mean_abs_dscore": 0.02005617693066597,
139
+ "mean_kl": 6.540709706121381e-06,
140
+ "n_gold": 2,
141
+ "nll_ref": 1.387121106483142,
142
+ "nll_cand": 1.3792866299819007,
143
+ "dnll": -0.00783447650124125,
144
+ "mean_abs_row_dnll": 0.00783447650124136,
145
+ "acc_ref": 0.5,
146
+ "acc_cand": 0.5,
147
+ "acc_drop_pp": 0.0
148
+ },
149
+ "multi_question": {
150
+ "n": 10,
151
+ "top1_agreement": 1.0,
152
+ "top1_flips": 0,
153
+ "max_abs_dscore": 0.06343865394592285,
154
+ "mean_abs_dscore": 0.018984448437889417,
155
+ "mean_kl": 0.00014264548450480086,
156
+ "n_gold": 10,
157
+ "nll_ref": 0.7606667410227165,
158
+ "nll_cand": 0.7742217565907421,
159
+ "dnll": 0.01355501556802563,
160
+ "mean_abs_row_dnll": 0.016602018627922027,
161
+ "acc_ref": 0.7,
162
+ "acc_cand": 0.7,
163
+ "acc_drop_pp": 0.0
164
+ },
165
+ "prefix_boundary": {
166
+ "n": 7,
167
+ "top1_agreement": 1.0,
168
+ "top1_flips": 0,
169
+ "max_abs_dscore": 0.07407116889953613,
170
+ "mean_abs_dscore": 0.02506211300690969,
171
+ "mean_kl": 0.000275167227242767,
172
+ "n_gold": 7,
173
+ "nll_ref": 0.9445710499909146,
174
+ "nll_cand": 0.9592024225354618,
175
+ "dnll": 0.014631372544547272,
176
+ "mean_abs_row_dnll": 0.026803385065686018,
177
+ "acc_ref": 0.5714285714285714,
178
+ "acc_cand": 0.5714285714285714,
179
+ "acc_drop_pp": 0.0
180
+ },
181
+ "qtype_choice": {
182
+ "n": 1,
183
+ "top1_agreement": 1.0,
184
+ "top1_flips": 0,
185
+ "max_abs_dscore": 0.039629340171813965,
186
+ "mean_abs_dscore": 0.01868373155593872,
187
+ "mean_kl": 9.758721027128357e-05,
188
+ "n_gold": 1,
189
+ "nll_ref": 1.2471588298811915,
190
+ "nll_cand": 1.2644290223938441,
191
+ "dnll": 0.017270192512652605,
192
+ "mean_abs_row_dnll": 0.017270192512652605,
193
+ "acc_ref": 0.0,
194
+ "acc_cand": 0.0,
195
+ "acc_drop_pp": 0.0
196
+ },
197
+ "qtype_noul": {
198
+ "n": 1,
199
+ "top1_agreement": 1.0,
200
+ "top1_flips": 0,
201
+ "max_abs_dscore": 0.04642695188522339,
202
+ "mean_abs_dscore": 0.02744242548942566,
203
+ "mean_kl": 0.00046207429765575576,
204
+ "n_gold": 1,
205
+ "nll_ref": 0.3643240477162341,
206
+ "nll_cand": 0.3445434412562239,
207
+ "dnll": -0.01978060646001023,
208
+ "mean_abs_row_dnll": 0.01978060646001023,
209
+ "acc_ref": 1.0,
210
+ "acc_cand": 1.0,
211
+ "acc_drop_pp": 0.0
212
+ },
213
+ "qtype_score": {
214
+ "n": 1,
215
+ "top1_agreement": 1.0,
216
+ "top1_flips": 0,
217
+ "max_abs_dscore": 0.04161477088928223,
218
+ "mean_abs_dscore": 0.015557724237442016,
219
+ "mean_kl": 1.4547309373278143e-05,
220
+ "n_gold": 1,
221
+ "nll_ref": 0.9393375234945696,
222
+ "nll_cand": 0.9372245712212975,
223
+ "dnll": -0.0021129522732720174,
224
+ "mean_abs_row_dnll": 0.0021129522732720174,
225
+ "acc_ref": 0.0,
226
+ "acc_cand": 0.0,
227
+ "acc_drop_pp": 0.0
228
+ },
229
+ "short_sb": {
230
+ "n": 6,
231
+ "top1_agreement": 1.0,
232
+ "top1_flips": 0,
233
+ "max_abs_dscore": 0.12482678890228271,
234
+ "mean_abs_dscore": 0.01820988009964663,
235
+ "mean_kl": 0.0005014431591724873,
236
+ "n_gold": 6,
237
+ "nll_ref": 0.11428482960768895,
238
+ "nll_cand": 0.123207743102961,
239
+ "dnll": 0.00892291349527205,
240
+ "mean_abs_row_dnll": 0.008996485578208901,
241
+ "acc_ref": 1.0,
242
+ "acc_cand": 1.0,
243
+ "acc_drop_pp": 0.0
244
+ }
245
+ },
246
+ "gates": {
247
+ "coverage_finite_tokens": {
248
+ "pass": true,
249
+ "problems": {},
250
+ "reference_missing_requests": 0
251
+ },
252
+ "top1": {
253
+ "threshold": 0.98,
254
+ "value": 1.0,
255
+ "pass": true
256
+ },
257
+ "dnll": {
258
+ "threshold": 0.02,
259
+ "value": 0.01638063705636561,
260
+ "pass": true
261
+ },
262
+ "acc_drop_pp": {
263
+ "threshold": 0.3,
264
+ "value": 0.0,
265
+ "resolution_pp": 2.3255813953488373,
266
+ "pass": null,
267
+ "note": "UNRESOLVED: 43 gold questions -> one flip = 2.33pp > 0.3pp; this fixture cannot resolve the gate (use the gate fixture with >= 334 rows; statistical power needs far more)"
268
+ }
269
+ },
270
+ "verdict": "INCONCLUSIVE",
271
+ "resolution_pp_per_flip": 2.3255813953488373,
272
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
273
+ "ref": "runs/macjev/runtime_v2_dev/reference_trained/ref_block_fp32.jsonl",
274
+ "cand": "runs/macjev/release_2b/parity/pred_mlx_affine8-g64_base.jsonl",
275
+ "fixture": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl"
276
+ }
validation/parity/compare_mlx_affine8-g64_gate_vs_block.json ADDED
@@ -0,0 +1,228 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "8bit",
3
+ "temperature": 0.8278650620942867,
4
+ "overall": {
5
+ "n": 1000,
6
+ "top1_agreement": 0.996,
7
+ "top1_flips": 4,
8
+ "max_abs_dscore": 0.3184394836425781,
9
+ "mean_abs_dscore": 0.016680597170951345,
10
+ "mean_kl": 0.0002134925530053613,
11
+ "n_gold": 1000,
12
+ "nll_ref": 0.4436679457518203,
13
+ "nll_cand": 0.44417151398689614,
14
+ "dnll": 0.0005035682350758575,
15
+ "mean_abs_row_dnll": 0.008527335296640175,
16
+ "acc_ref": 0.808,
17
+ "acc_cand": 0.808,
18
+ "acc_drop_pp": 0.0
19
+ },
20
+ "per_category": {
21
+ "di_knowledge": {
22
+ "n": 93,
23
+ "top1_agreement": 0.978494623655914,
24
+ "top1_flips": 2,
25
+ "max_abs_dscore": 0.10516834259033203,
26
+ "mean_abs_dscore": 0.01826677256272753,
27
+ "mean_kl": 0.0001961899290394059,
28
+ "n_gold": 93,
29
+ "nll_ref": 0.7907936240501159,
30
+ "nll_cand": 0.7950259330695674,
31
+ "dnll": 0.004232309019451486,
32
+ "mean_abs_row_dnll": 0.012335049730692368,
33
+ "acc_ref": 0.6559139784946236,
34
+ "acc_cand": 0.6559139784946236,
35
+ "acc_drop_pp": 0.0
36
+ },
37
+ "di_language": {
38
+ "n": 93,
39
+ "top1_agreement": 1.0,
40
+ "top1_flips": 0,
41
+ "max_abs_dscore": 0.12582671642303467,
42
+ "mean_abs_dscore": 0.019231186946019475,
43
+ "mean_kl": 0.0002764365455637818,
44
+ "n_gold": 93,
45
+ "nll_ref": 0.43828192127009435,
46
+ "nll_cand": 0.4350339331964042,
47
+ "dnll": -0.0032479880736901445,
48
+ "mean_abs_row_dnll": 0.01105854314657161,
49
+ "acc_ref": 0.8494623655913979,
50
+ "acc_cand": 0.8494623655913979,
51
+ "acc_drop_pp": 0.0
52
+ },
53
+ "di_retrieval": {
54
+ "n": 93,
55
+ "top1_agreement": 1.0,
56
+ "top1_flips": 0,
57
+ "max_abs_dscore": 0.16489863395690918,
58
+ "mean_abs_dscore": 0.013032539303739148,
59
+ "mean_kl": 9.17728737128296e-05,
60
+ "n_gold": 93,
61
+ "nll_ref": 0.07565367364179643,
62
+ "nll_cand": 0.07596664403414172,
63
+ "dnll": 0.0003129703923452909,
64
+ "mean_abs_row_dnll": 0.0027333661229381077,
65
+ "acc_ref": 0.978494623655914,
66
+ "acc_cand": 0.978494623655914,
67
+ "acc_drop_pp": 0.0
68
+ },
69
+ "format": {
70
+ "n": 75,
71
+ "top1_agreement": 1.0,
72
+ "top1_flips": 0,
73
+ "max_abs_dscore": 0.15371152758598328,
74
+ "mean_abs_dscore": 0.014854268128133183,
75
+ "mean_kl": 8.706661370645814e-05,
76
+ "n_gold": 75,
77
+ "nll_ref": 0.09087868406020969,
78
+ "nll_cand": 0.09241995785860305,
79
+ "dnll": 0.001541273798393361,
80
+ "mean_abs_row_dnll": 0.003729830247715702,
81
+ "acc_ref": 0.9733333333333334,
82
+ "acc_cand": 0.9733333333333334,
83
+ "acc_drop_pp": 0.0
84
+ },
85
+ "general": {
86
+ "n": 93,
87
+ "top1_agreement": 1.0,
88
+ "top1_flips": 0,
89
+ "max_abs_dscore": 0.1003103256225586,
90
+ "mean_abs_dscore": 0.01342117658100456,
91
+ "mean_kl": 7.21948109281326e-05,
92
+ "n_gold": 93,
93
+ "nll_ref": 0.13061020512422436,
94
+ "nll_cand": 0.12993521810236933,
95
+ "dnll": -0.0006749870218550336,
96
+ "mean_abs_row_dnll": 0.003786342376918514,
97
+ "acc_ref": 0.967741935483871,
98
+ "acc_cand": 0.967741935483871,
99
+ "acc_drop_pp": 0.0
100
+ },
101
+ "hard_short": {
102
+ "n": 93,
103
+ "top1_agreement": 0.989247311827957,
104
+ "top1_flips": 1,
105
+ "max_abs_dscore": 0.12274712324142456,
106
+ "mean_abs_dscore": 0.020271506775084727,
107
+ "mean_kl": 0.00027143138298962177,
108
+ "n_gold": 93,
109
+ "nll_ref": 0.7835153258109905,
110
+ "nll_cand": 0.7861608190184016,
111
+ "dnll": 0.0026454932074111426,
112
+ "mean_abs_row_dnll": 0.014292107382686168,
113
+ "acc_ref": 0.6021505376344086,
114
+ "acc_cand": 0.5913978494623656,
115
+ "acc_drop_pp": 1.0752688172043001
116
+ },
117
+ "long": {
118
+ "n": 92,
119
+ "top1_agreement": 1.0,
120
+ "top1_flips": 0,
121
+ "max_abs_dscore": 0.3184394836425781,
122
+ "mean_abs_dscore": 0.020006766643251753,
123
+ "mean_kl": 0.0007467326030115854,
124
+ "n_gold": 92,
125
+ "nll_ref": 0.10827285500173092,
126
+ "nll_cand": 0.10447898892584594,
127
+ "dnll": -0.0037938660758849857,
128
+ "mean_abs_row_dnll": 0.008173341485227314,
129
+ "acc_ref": 0.967391304347826,
130
+ "acc_cand": 0.967391304347826,
131
+ "acc_drop_pp": 0.0
132
+ },
133
+ "mac": {
134
+ "n": 92,
135
+ "top1_agreement": 1.0,
136
+ "top1_flips": 0,
137
+ "max_abs_dscore": 0.05875861644744873,
138
+ "mean_abs_dscore": 0.013405789823635765,
139
+ "mean_kl": 3.8362254891210755e-05,
140
+ "n_gold": 92,
141
+ "nll_ref": 0.3262944821095264,
142
+ "nll_cand": 0.3269720723076003,
143
+ "dnll": 0.0006775901980738963,
144
+ "mean_abs_row_dnll": 0.0040734288260641455,
145
+ "acc_ref": 0.8369565217391305,
146
+ "acc_cand": 0.8369565217391305,
147
+ "acc_drop_pp": 0.0
148
+ },
149
+ "public_ingests": {
150
+ "n": 92,
151
+ "top1_agreement": 0.9891304347826086,
152
+ "top1_flips": 1,
153
+ "max_abs_dscore": 0.146925687789917,
154
+ "mean_abs_dscore": 0.020489049231511028,
155
+ "mean_kl": 0.00030913724098080613,
156
+ "n_gold": 92,
157
+ "nll_ref": 0.7667467143475788,
158
+ "nll_cand": 0.7701994879203283,
159
+ "dnll": 0.0034527735727495346,
160
+ "mean_abs_row_dnll": 0.016463952376233926,
161
+ "acc_ref": 0.6521739130434783,
162
+ "acc_cand": 0.6630434782608695,
163
+ "acc_drop_pp": -1.0869565217391242
164
+ },
165
+ "themes": {
166
+ "n": 92,
167
+ "top1_agreement": 1.0,
168
+ "top1_flips": 0,
169
+ "max_abs_dscore": 0.05321455001831055,
170
+ "mean_abs_dscore": 0.012158645316958427,
171
+ "mean_kl": 2.255066396046878e-05,
172
+ "n_gold": 92,
173
+ "nll_ref": 0.27025191463643466,
174
+ "nll_cand": 0.27031868131249703,
175
+ "dnll": 6.676667606236864e-05,
176
+ "mean_abs_row_dnll": 0.003036639894596646,
177
+ "acc_ref": 0.8695652173913043,
178
+ "acc_cand": 0.8695652173913043,
179
+ "acc_drop_pp": 0.0
180
+ },
181
+ "typed_synthetic": {
182
+ "n": 92,
183
+ "top1_agreement": 1.0,
184
+ "top1_flips": 0,
185
+ "max_abs_dscore": 0.07491475343704224,
186
+ "mean_abs_dscore": 0.018002478546206502,
187
+ "mean_kl": 0.00021491486269545232,
188
+ "n_gold": 92,
189
+ "nll_ref": 1.0338530850662833,
190
+ "nll_cand": 1.0343635982006703,
191
+ "dnll": 0.0005105131343869918,
192
+ "mean_abs_row_dnll": 0.0132145397374374,
193
+ "acc_ref": 0.5652173913043478,
194
+ "acc_cand": 0.5652173913043478,
195
+ "acc_drop_pp": 0.0
196
+ }
197
+ },
198
+ "gates": {
199
+ "coverage_finite_tokens": {
200
+ "pass": true,
201
+ "problems": {},
202
+ "reference_missing_requests": 0
203
+ },
204
+ "top1": {
205
+ "threshold": 0.98,
206
+ "value": 0.996,
207
+ "pass": true
208
+ },
209
+ "dnll": {
210
+ "threshold": 0.02,
211
+ "value": 0.0005035682350758575,
212
+ "pass": true
213
+ },
214
+ "acc_drop_pp": {
215
+ "threshold": 0.3,
216
+ "value": 0.0,
217
+ "resolution_pp": 0.1,
218
+ "pass": true,
219
+ "note": null
220
+ }
221
+ },
222
+ "verdict": "PASS",
223
+ "resolution_pp_per_flip": 0.1,
224
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
225
+ "ref": "runs/macjev/runtime_v2_dev/reference_trained/gate_ref_block_fp32.jsonl",
226
+ "cand": "runs/macjev/release_2b/parity/pred_mlx_affine8-g64_gate.jsonl",
227
+ "fixture": "runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl"
228
+ }
validation/parity/compare_mlx_bf16_base_vs_block.json ADDED
@@ -0,0 +1,269 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "bf16",
3
+ "temperature": 0.8278650620942867,
4
+ "overall": {
5
+ "n": 43,
6
+ "top1_agreement": 1.0,
7
+ "top1_flips": 0,
8
+ "max_abs_dscore": 0.08836114406585693,
9
+ "mean_abs_dscore": 0.01365492056503762,
10
+ "mean_kl": 9.883488035170103e-05,
11
+ "n_gold": 43,
12
+ "nll_ref": 0.670433307194272,
13
+ "nll_cand": 0.6677389810942516,
14
+ "dnll": -0.0026943261000204055,
15
+ "mean_abs_row_dnll": 0.009347213165255148,
16
+ "acc_ref": 0.7209302325581395,
17
+ "acc_cand": 0.7209302325581395,
18
+ "acc_drop_pp": 0.0
19
+ },
20
+ "per_category": {
21
+ "catalogue_overflow": {
22
+ "n": 3,
23
+ "top1_agreement": 1.0,
24
+ "top1_flips": 0,
25
+ "max_abs_dscore": 0.08836114406585693,
26
+ "mean_abs_dscore": 0.012576654652096578,
27
+ "mean_kl": 9.049678076068058e-05,
28
+ "n_gold": 3,
29
+ "nll_ref": 0.5023960827418966,
30
+ "nll_cand": 0.49688457645611867,
31
+ "dnll": -0.005511506285777956,
32
+ "mean_abs_row_dnll": 0.005740943453378988,
33
+ "acc_ref": 0.6666666666666666,
34
+ "acc_cand": 0.6666666666666666,
35
+ "acc_drop_pp": 0.0
36
+ },
37
+ "catalogue_overflow_long_state": {
38
+ "n": 1,
39
+ "top1_agreement": 1.0,
40
+ "top1_flips": 0,
41
+ "max_abs_dscore": 0.04151296615600586,
42
+ "mean_abs_dscore": 0.012695910914844235,
43
+ "mean_kl": 0.00019275121223053925,
44
+ "n_gold": 1,
45
+ "nll_ref": 1.868953922917252,
46
+ "nll_cand": 1.8621515972563385,
47
+ "dnll": -0.006802325660913544,
48
+ "mean_abs_row_dnll": 0.006802325660913544,
49
+ "acc_ref": 0.0,
50
+ "acc_cand": 0.0,
51
+ "acc_drop_pp": 0.0
52
+ },
53
+ "long_16k": {
54
+ "n": 3,
55
+ "top1_agreement": 1.0,
56
+ "top1_flips": 0,
57
+ "max_abs_dscore": 0.05213227868080139,
58
+ "mean_abs_dscore": 0.014879310828434184,
59
+ "mean_kl": 0.000149667886473795,
60
+ "n_gold": 3,
61
+ "nll_ref": 1.098996986024778,
62
+ "nll_cand": 1.0956083262917284,
63
+ "dnll": -0.0033886597330494705,
64
+ "mean_abs_row_dnll": 0.006144886758009865,
65
+ "acc_ref": 0.6666666666666666,
66
+ "acc_cand": 0.6666666666666666,
67
+ "acc_drop_pp": 0.0
68
+ },
69
+ "long_24k": {
70
+ "n": 2,
71
+ "top1_agreement": 1.0,
72
+ "top1_flips": 0,
73
+ "max_abs_dscore": 0.03363656997680664,
74
+ "mean_abs_dscore": 0.012938144306341806,
75
+ "mean_kl": 0.0001715685914165032,
76
+ "n_gold": 2,
77
+ "nll_ref": 0.43411292245601785,
78
+ "nll_cand": 0.42479629433910976,
79
+ "dnll": -0.009316628116908088,
80
+ "mean_abs_row_dnll": 0.00931662811690806,
81
+ "acc_ref": 1.0,
82
+ "acc_cand": 1.0,
83
+ "acc_drop_pp": 0.0
84
+ },
85
+ "many_options": {
86
+ "n": 2,
87
+ "top1_agreement": 1.0,
88
+ "top1_flips": 0,
89
+ "max_abs_dscore": 0.03784656524658203,
90
+ "mean_abs_dscore": 0.0131730338682731,
91
+ "mean_kl": 4.283448514828303e-06,
92
+ "n_gold": 2,
93
+ "nll_ref": 0.012124599411053285,
94
+ "nll_cand": 0.01233819862308103,
95
+ "dnll": 0.0002135992120277444,
96
+ "mean_abs_row_dnll": 0.00021359921202774614,
97
+ "acc_ref": 1.0,
98
+ "acc_cand": 1.0,
99
+ "acc_drop_pp": 0.0
100
+ },
101
+ "multi_block_3k": {
102
+ "n": 2,
103
+ "top1_agreement": 1.0,
104
+ "top1_flips": 0,
105
+ "max_abs_dscore": 0.053168416023254395,
106
+ "mean_abs_dscore": 0.011156398057937621,
107
+ "mean_kl": 0.00020998625581104233,
108
+ "n_gold": 2,
109
+ "nll_ref": 0.10264503770560551,
110
+ "nll_cand": 0.09699945875175084,
111
+ "dnll": -0.005645578953854674,
112
+ "mean_abs_row_dnll": 0.005792563090663002,
113
+ "acc_ref": 1.0,
114
+ "acc_cand": 1.0,
115
+ "acc_drop_pp": 0.0
116
+ },
117
+ "multi_block_5k": {
118
+ "n": 2,
119
+ "top1_agreement": 1.0,
120
+ "top1_flips": 0,
121
+ "max_abs_dscore": 0.018996506929397583,
122
+ "mean_abs_dscore": 0.010537676513195038,
123
+ "mean_kl": 7.5737981061909e-05,
124
+ "n_gold": 2,
125
+ "nll_ref": 0.41414880456154163,
126
+ "nll_cand": 0.40761999179423464,
127
+ "dnll": -0.006528812767306991,
128
+ "mean_abs_row_dnll": 0.0065288127673070045,
129
+ "acc_ref": 1.0,
130
+ "acc_cand": 1.0,
131
+ "acc_drop_pp": 0.0
132
+ },
133
+ "multi_block_9k": {
134
+ "n": 2,
135
+ "top1_agreement": 1.0,
136
+ "top1_flips": 0,
137
+ "max_abs_dscore": 0.033649444580078125,
138
+ "mean_abs_dscore": 0.00982309877872467,
139
+ "mean_kl": 8.772959532626077e-05,
140
+ "n_gold": 2,
141
+ "nll_ref": 1.387121106483142,
142
+ "nll_cand": 1.389387240331538,
143
+ "dnll": 0.0022661338483960236,
144
+ "mean_abs_row_dnll": 0.009883718382939666,
145
+ "acc_ref": 0.5,
146
+ "acc_cand": 0.5,
147
+ "acc_drop_pp": 0.0
148
+ },
149
+ "multi_question": {
150
+ "n": 10,
151
+ "top1_agreement": 1.0,
152
+ "top1_flips": 0,
153
+ "max_abs_dscore": 0.06624865531921387,
154
+ "mean_abs_dscore": 0.014526768724123637,
155
+ "mean_kl": 8.529135561399713e-05,
156
+ "n_gold": 10,
157
+ "nll_ref": 0.7606667410227165,
158
+ "nll_cand": 0.7570780864187747,
159
+ "dnll": -0.0035886546039417544,
160
+ "mean_abs_row_dnll": 0.010879836749203488,
161
+ "acc_ref": 0.7,
162
+ "acc_cand": 0.7,
163
+ "acc_drop_pp": 0.0
164
+ },
165
+ "prefix_boundary": {
166
+ "n": 7,
167
+ "top1_agreement": 1.0,
168
+ "top1_flips": 0,
169
+ "max_abs_dscore": 0.047842979431152344,
170
+ "mean_abs_dscore": 0.014937927183650791,
171
+ "mean_kl": 0.00011731551621384146,
172
+ "n_gold": 7,
173
+ "nll_ref": 0.9445710499909146,
174
+ "nll_cand": 0.9344505690212561,
175
+ "dnll": -0.010120480969658452,
176
+ "mean_abs_row_dnll": 0.01758201360584486,
177
+ "acc_ref": 0.5714285714285714,
178
+ "acc_cand": 0.5714285714285714,
179
+ "acc_drop_pp": 0.0
180
+ },
181
+ "qtype_choice": {
182
+ "n": 1,
183
+ "top1_agreement": 1.0,
184
+ "top1_flips": 0,
185
+ "max_abs_dscore": 0.02133503556251526,
186
+ "mean_abs_dscore": 0.013152092695236206,
187
+ "mean_kl": 8.789457820938037e-05,
188
+ "n_gold": 1,
189
+ "nll_ref": 1.2471588298811915,
190
+ "nll_cand": 1.267813737179954,
191
+ "dnll": 0.020654907298762515,
192
+ "mean_abs_row_dnll": 0.020654907298762515,
193
+ "acc_ref": 0.0,
194
+ "acc_cand": 0.0,
195
+ "acc_drop_pp": 0.0
196
+ },
197
+ "qtype_noul": {
198
+ "n": 1,
199
+ "top1_agreement": 1.0,
200
+ "top1_flips": 0,
201
+ "max_abs_dscore": 0.010687410831451416,
202
+ "mean_abs_dscore": 0.006474539637565613,
203
+ "mean_kl": 2.599908743399675e-05,
204
+ "n_gold": 1,
205
+ "nll_ref": 0.3643240477162341,
206
+ "nll_cand": 0.3691259380231449,
207
+ "dnll": 0.004801890306910805,
208
+ "mean_abs_row_dnll": 0.004801890306910805,
209
+ "acc_ref": 1.0,
210
+ "acc_cand": 1.0,
211
+ "acc_drop_pp": 0.0
212
+ },
213
+ "qtype_score": {
214
+ "n": 1,
215
+ "top1_agreement": 1.0,
216
+ "top1_flips": 0,
217
+ "max_abs_dscore": 0.03596752882003784,
218
+ "mean_abs_dscore": 0.013961112499237061,
219
+ "mean_kl": 0.0002653113090994117,
220
+ "n_gold": 1,
221
+ "nll_ref": 0.9393375234945696,
222
+ "nll_cand": 0.9660497441698873,
223
+ "dnll": 0.026712220675317755,
224
+ "mean_abs_row_dnll": 0.026712220675317755,
225
+ "acc_ref": 0.0,
226
+ "acc_cand": 0.0,
227
+ "acc_drop_pp": 0.0
228
+ },
229
+ "short_sb": {
230
+ "n": 6,
231
+ "top1_agreement": 1.0,
232
+ "top1_flips": 0,
233
+ "max_abs_dscore": 0.04204094409942627,
234
+ "mean_abs_dscore": 0.015570025255400981,
235
+ "mean_kl": 3.078595875807344e-05,
236
+ "n_gold": 6,
237
+ "nll_ref": 0.11428482960768895,
238
+ "nll_cand": 0.11598987452733013,
239
+ "dnll": 0.0017050449196411854,
240
+ "mean_abs_row_dnll": 0.0019930376095433936,
241
+ "acc_ref": 1.0,
242
+ "acc_cand": 1.0,
243
+ "acc_drop_pp": 0.0
244
+ }
245
+ },
246
+ "gates": {
247
+ "coverage_finite_tokens": {
248
+ "pass": true,
249
+ "problems": {},
250
+ "reference_missing_requests": 0
251
+ },
252
+ "top1": {
253
+ "threshold": 0.99,
254
+ "value": 1.0,
255
+ "pass": true
256
+ },
257
+ "dnll": {
258
+ "threshold": 0.02,
259
+ "value": -0.0026943261000204055,
260
+ "pass": true
261
+ }
262
+ },
263
+ "verdict": "PASS",
264
+ "resolution_pp_per_flip": 2.3255813953488373,
265
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
266
+ "ref": "runs/macjev/runtime_v2_dev/reference_trained/ref_block_fp32.jsonl",
267
+ "cand": "runs/macjev/release_2b/parity/pred_mlx_bf16_base.jsonl",
268
+ "fixture": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl"
269
+ }
validation/parity/compare_mlx_bf16_gate_vs_block.json ADDED
@@ -0,0 +1,221 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "bf16",
3
+ "temperature": 0.8278650620942867,
4
+ "overall": {
5
+ "n": 1000,
6
+ "top1_agreement": 0.997,
7
+ "top1_flips": 3,
8
+ "max_abs_dscore": 0.13961565494537354,
9
+ "mean_abs_dscore": 0.012822698334419146,
10
+ "mean_kl": 7.013363207503902e-05,
11
+ "n_gold": 1000,
12
+ "nll_ref": 0.4436679457518203,
13
+ "nll_cand": 0.4437944677940992,
14
+ "dnll": 0.0001265220422789204,
15
+ "mean_abs_row_dnll": 0.005641898326944814,
16
+ "acc_ref": 0.808,
17
+ "acc_cand": 0.805,
18
+ "acc_drop_pp": 0.30000000000000027
19
+ },
20
+ "per_category": {
21
+ "di_knowledge": {
22
+ "n": 93,
23
+ "top1_agreement": 1.0,
24
+ "top1_flips": 0,
25
+ "max_abs_dscore": 0.07215514779090881,
26
+ "mean_abs_dscore": 0.014889913475799058,
27
+ "mean_kl": 0.00013722336018850683,
28
+ "n_gold": 93,
29
+ "nll_ref": 0.7907936240501159,
30
+ "nll_cand": 0.7927041608827371,
31
+ "dnll": 0.0019105368326212124,
32
+ "mean_abs_row_dnll": 0.01039648465267476,
33
+ "acc_ref": 0.6559139784946236,
34
+ "acc_cand": 0.6559139784946236,
35
+ "acc_drop_pp": 0.0
36
+ },
37
+ "di_language": {
38
+ "n": 93,
39
+ "top1_agreement": 1.0,
40
+ "top1_flips": 0,
41
+ "max_abs_dscore": 0.0584871768951416,
42
+ "mean_abs_dscore": 0.013199496934170362,
43
+ "mean_kl": 4.376886766709004e-05,
44
+ "n_gold": 93,
45
+ "nll_ref": 0.43828192127009435,
46
+ "nll_cand": 0.43613660599650544,
47
+ "dnll": -0.0021453152735889103,
48
+ "mean_abs_row_dnll": 0.005769073926221885,
49
+ "acc_ref": 0.8494623655913979,
50
+ "acc_cand": 0.8494623655913979,
51
+ "acc_drop_pp": 0.0
52
+ },
53
+ "di_retrieval": {
54
+ "n": 93,
55
+ "top1_agreement": 1.0,
56
+ "top1_flips": 0,
57
+ "max_abs_dscore": 0.13961565494537354,
58
+ "mean_abs_dscore": 0.011278054842327665,
59
+ "mean_kl": 6.320183776936612e-05,
60
+ "n_gold": 93,
61
+ "nll_ref": 0.07565367364179643,
62
+ "nll_cand": 0.07529180302084762,
63
+ "dnll": -0.0003618706209488065,
64
+ "mean_abs_row_dnll": 0.0018865181948961214,
65
+ "acc_ref": 0.978494623655914,
66
+ "acc_cand": 0.978494623655914,
67
+ "acc_drop_pp": 0.0
68
+ },
69
+ "format": {
70
+ "n": 75,
71
+ "top1_agreement": 1.0,
72
+ "top1_flips": 0,
73
+ "max_abs_dscore": 0.07049310207366943,
74
+ "mean_abs_dscore": 0.012186950511526283,
75
+ "mean_kl": 1.866440251394625e-05,
76
+ "n_gold": 75,
77
+ "nll_ref": 0.09087868406020969,
78
+ "nll_cand": 0.09027726831006917,
79
+ "dnll": -0.0006014157501405132,
80
+ "mean_abs_row_dnll": 0.0017343322558420642,
81
+ "acc_ref": 0.9733333333333334,
82
+ "acc_cand": 0.9733333333333334,
83
+ "acc_drop_pp": 0.0
84
+ },
85
+ "general": {
86
+ "n": 93,
87
+ "top1_agreement": 1.0,
88
+ "top1_flips": 0,
89
+ "max_abs_dscore": 0.053872108459472656,
90
+ "mean_abs_dscore": 0.011199172480311083,
91
+ "mean_kl": 8.964417186979084e-06,
92
+ "n_gold": 93,
93
+ "nll_ref": 0.13061020512422436,
94
+ "nll_cand": 0.12994049456566417,
95
+ "dnll": -0.0006697105585601881,
96
+ "mean_abs_row_dnll": 0.0015611860129022736,
97
+ "acc_ref": 0.967741935483871,
98
+ "acc_cand": 0.967741935483871,
99
+ "acc_drop_pp": 0.0
100
+ },
101
+ "hard_short": {
102
+ "n": 93,
103
+ "top1_agreement": 0.989247311827957,
104
+ "top1_flips": 1,
105
+ "max_abs_dscore": 0.06802543997764587,
106
+ "mean_abs_dscore": 0.014637837174057167,
107
+ "mean_kl": 0.00013160361749015227,
108
+ "n_gold": 93,
109
+ "nll_ref": 0.7835153258109905,
110
+ "nll_cand": 0.782783474188176,
111
+ "dnll": -0.0007318516228145278,
112
+ "mean_abs_row_dnll": 0.01093632719811279,
113
+ "acc_ref": 0.6021505376344086,
114
+ "acc_cand": 0.5913978494623656,
115
+ "acc_drop_pp": 1.0752688172043001
116
+ },
117
+ "long": {
118
+ "n": 92,
119
+ "top1_agreement": 1.0,
120
+ "top1_flips": 0,
121
+ "max_abs_dscore": 0.1342533826828003,
122
+ "mean_abs_dscore": 0.013823732117665294,
123
+ "mean_kl": 6.168921301697641e-05,
124
+ "n_gold": 92,
125
+ "nll_ref": 0.10827285500173092,
126
+ "nll_cand": 0.10768749311257943,
127
+ "dnll": -0.000585361889151495,
128
+ "mean_abs_row_dnll": 0.002564256755867923,
129
+ "acc_ref": 0.967391304347826,
130
+ "acc_cand": 0.967391304347826,
131
+ "acc_drop_pp": 0.0
132
+ },
133
+ "mac": {
134
+ "n": 92,
135
+ "top1_agreement": 1.0,
136
+ "top1_flips": 0,
137
+ "max_abs_dscore": 0.04618799686431885,
138
+ "mean_abs_dscore": 0.01143809092109618,
139
+ "mean_kl": 2.6581458539860547e-05,
140
+ "n_gold": 92,
141
+ "nll_ref": 0.3262944821095264,
142
+ "nll_cand": 0.3269180887247012,
143
+ "dnll": 0.0006236066151747988,
144
+ "mean_abs_row_dnll": 0.003772370527716574,
145
+ "acc_ref": 0.8369565217391305,
146
+ "acc_cand": 0.8369565217391305,
147
+ "acc_drop_pp": 0.0
148
+ },
149
+ "public_ingests": {
150
+ "n": 92,
151
+ "top1_agreement": 1.0,
152
+ "top1_flips": 0,
153
+ "max_abs_dscore": 0.10585364699363708,
154
+ "mean_abs_dscore": 0.015097456673118565,
155
+ "mean_kl": 0.00015103258358910773,
156
+ "n_gold": 92,
157
+ "nll_ref": 0.7667467143475788,
158
+ "nll_cand": 0.7682457652594957,
159
+ "dnll": 0.0014990509119169326,
160
+ "mean_abs_row_dnll": 0.010491535553020102,
161
+ "acc_ref": 0.6521739130434783,
162
+ "acc_cand": 0.6521739130434783,
163
+ "acc_drop_pp": 0.0
164
+ },
165
+ "themes": {
166
+ "n": 92,
167
+ "top1_agreement": 0.9891304347826086,
168
+ "top1_flips": 1,
169
+ "max_abs_dscore": 0.04355478286743164,
170
+ "mean_abs_dscore": 0.010605669458923132,
171
+ "mean_kl": 2.4907957260022938e-05,
172
+ "n_gold": 92,
173
+ "nll_ref": 0.27025191463643466,
174
+ "nll_cand": 0.2710970978349834,
175
+ "dnll": 0.0008451831985487601,
176
+ "mean_abs_row_dnll": 0.0033373064356656897,
177
+ "acc_ref": 0.8695652173913043,
178
+ "acc_cand": 0.8586956521739131,
179
+ "acc_drop_pp": 1.0869565217391242
180
+ },
181
+ "typed_synthetic": {
182
+ "n": 92,
183
+ "top1_agreement": 0.9891304347826086,
184
+ "top1_flips": 1,
185
+ "max_abs_dscore": 0.04864954948425293,
186
+ "mean_abs_dscore": 0.01256397343500986,
187
+ "mean_kl": 9.395103279401365e-05,
188
+ "n_gold": 92,
189
+ "nll_ref": 1.0338530850662833,
190
+ "nll_cand": 1.0353560613294195,
191
+ "dnll": 0.0015029762631362242,
192
+ "mean_abs_row_dnll": 0.008864003979572451,
193
+ "acc_ref": 0.5652173913043478,
194
+ "acc_cand": 0.5543478260869565,
195
+ "acc_drop_pp": 1.0869565217391242
196
+ }
197
+ },
198
+ "gates": {
199
+ "coverage_finite_tokens": {
200
+ "pass": true,
201
+ "problems": {},
202
+ "reference_missing_requests": 0
203
+ },
204
+ "top1": {
205
+ "threshold": 0.99,
206
+ "value": 0.997,
207
+ "pass": true
208
+ },
209
+ "dnll": {
210
+ "threshold": 0.02,
211
+ "value": 0.0001265220422789204,
212
+ "pass": true
213
+ }
214
+ },
215
+ "verdict": "PASS",
216
+ "resolution_pp_per_flip": 0.1,
217
+ "note": "small fixtures cannot resolve 0.3pp accuracy gates; accuracy on a parity fixture is only indicative (synthetic long rows carry the base row's gold label).",
218
+ "ref": "runs/macjev/runtime_v2_dev/reference_trained/gate_ref_block_fp32.jsonl",
219
+ "cand": "runs/macjev/release_2b/parity/pred_mlx_bf16_gate.jsonl",
220
+ "fixture": "runs/macjev/runtime_v2_dev/reference/gate_fixture_n1000.jsonl"
221
+ }
validation/parity/cross_format_dp.json ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base:torch_fp32_cpu": {
3
+ "n": 43,
4
+ "top1_agree": 1.0,
5
+ "max_abs_dp": 4.782633115096857e-06,
6
+ "mean_max_abs_dp": 6.972375795818997e-07
7
+ },
8
+ "base:gguf_f16": {
9
+ "n": 43,
10
+ "top1_agree": 1.0,
11
+ "max_abs_dp": 0.0006297677378335198,
12
+ "mean_max_abs_dp": 0.00017737997866918594
13
+ },
14
+ "base:gguf_q8_0": {
15
+ "n": 43,
16
+ "top1_agree": 1.0,
17
+ "max_abs_dp": 0.021289327356281362,
18
+ "mean_max_abs_dp": 0.0044768677260933745
19
+ },
20
+ "base:gguf_q4_k_m": {
21
+ "n": 43,
22
+ "top1_agree": 0.9767441860465116,
23
+ "max_abs_dp": 0.15778925110927658,
24
+ "mean_max_abs_dp": 0.047022506153486646
25
+ },
26
+ "base:mlx_bf16": {
27
+ "n": 43,
28
+ "top1_agree": 1.0,
29
+ "max_abs_dp": 0.010765431767972844,
30
+ "mean_max_abs_dp": 0.003890125022245109
31
+ },
32
+ "base:mlx_8bit": {
33
+ "n": 43,
34
+ "top1_agree": 1.0,
35
+ "max_abs_dp": 0.04628699142360499,
36
+ "mean_max_abs_dp": 0.007026009064181663
37
+ },
38
+ "gate:gguf_f16": {
39
+ "n": 1000,
40
+ "top1_agree": 1.0,
41
+ "max_abs_dp": 0.0013728371250730786,
42
+ "mean_max_abs_dp": 0.00012138780447113503
43
+ },
44
+ "gate:gguf_q8_0": {
45
+ "n": 1000,
46
+ "top1_agree": 0.997,
47
+ "max_abs_dp": 0.033479594816581415,
48
+ "mean_max_abs_dp": 0.0020247272188215755
49
+ },
50
+ "gate:gguf_q4_k_m": {
51
+ "n": 1000,
52
+ "top1_agree": 0.957,
53
+ "max_abs_dp": 0.34599917206983777,
54
+ "mean_max_abs_dp": 0.02317308476687987
55
+ },
56
+ "gate:mlx_bf16": {
57
+ "n": 1000,
58
+ "top1_agree": 0.997,
59
+ "max_abs_dp": 0.03549758339084519,
60
+ "mean_max_abs_dp": 0.0025865935031305484
61
+ },
62
+ "gate:mlx_8bit": {
63
+ "n": 1000,
64
+ "top1_agree": 0.996,
65
+ "max_abs_dp": 0.1623697296878569,
66
+ "mean_max_abs_dp": 0.0039011049015487734
67
+ }
68
+ }
validation/runtime/fixture_verify_8bit.json ADDED
@@ -0,0 +1,319 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "precision": "8bit",
3
+ "dev_seconds": 1324.8,
4
+ "default_T": 0.8278650620942867,
5
+ "fixture_requests": 35,
6
+ "fixture_questions": 43,
7
+ "fixture_tokens": 214700,
8
+ "overflow_q": 5,
9
+ "max_abs_dscore": 0.0,
10
+ "max_abs_dprob_vs_softmax_dev_over_T": 0.0,
11
+ "text_e2e_max_abs_dscore_vs_dev": 0.0,
12
+ "text_ids_equal_dev": true,
13
+ "score_many_vs_separate_fresh_decide_max_abs": 0.0,
14
+ "order_invariance_max_abs": 0.0,
15
+ "runtime_seconds": 286.5,
16
+ "per_question": [
17
+ [
18
+ "multi_question-0",
19
+ 0,
20
+ 450,
21
+ 5,
22
+ 0.0
23
+ ],
24
+ [
25
+ "multi_question-0",
26
+ 1,
27
+ 364,
28
+ 4,
29
+ 0.0
30
+ ],
31
+ [
32
+ "multi_question-0",
33
+ 2,
34
+ 356,
35
+ 2,
36
+ 0.0
37
+ ],
38
+ [
39
+ "multi_question-0",
40
+ 3,
41
+ 345,
42
+ 4,
43
+ 0.0
44
+ ],
45
+ [
46
+ "multi_question-0",
47
+ 4,
48
+ 319,
49
+ 2,
50
+ 0.0
51
+ ],
52
+ [
53
+ "multi_question-1",
54
+ 0,
55
+ 5416,
56
+ 2,
57
+ 0.0
58
+ ],
59
+ [
60
+ "multi_question-1",
61
+ 1,
62
+ 5471,
63
+ 2,
64
+ 0.0
65
+ ],
66
+ [
67
+ "multi_question-1",
68
+ 2,
69
+ 5425,
70
+ 2,
71
+ 0.0
72
+ ],
73
+ [
74
+ "multi_question-1",
75
+ 3,
76
+ 5517,
77
+ 6,
78
+ 0.0
79
+ ],
80
+ [
81
+ "multi_question-1",
82
+ 4,
83
+ 5428,
84
+ 2,
85
+ 0.0
86
+ ],
87
+ [
88
+ "short_sb-themes-0",
89
+ 0,
90
+ 105,
91
+ 2,
92
+ 0.0
93
+ ],
94
+ [
95
+ "short_sb-public_ingests-0",
96
+ 0,
97
+ 382,
98
+ 16,
99
+ 0.0
100
+ ],
101
+ [
102
+ "short_sb-di_retrieval-0",
103
+ 0,
104
+ 197,
105
+ 2,
106
+ 0.0
107
+ ],
108
+ [
109
+ "short_sb-di_knowledge-0",
110
+ 0,
111
+ 72,
112
+ 4,
113
+ 0.0
114
+ ],
115
+ [
116
+ "short_sb-hard_short-0",
117
+ 0,
118
+ 517,
119
+ 3,
120
+ 0.0
121
+ ],
122
+ [
123
+ "short_sb-format-0",
124
+ 0,
125
+ 166,
126
+ 21,
127
+ 0.0
128
+ ],
129
+ [
130
+ "qtype_choice-0",
131
+ 0,
132
+ 594,
133
+ 3,
134
+ 0.0
135
+ ],
136
+ [
137
+ "qtype_noul-0",
138
+ 0,
139
+ 246,
140
+ 2,
141
+ 0.0
142
+ ],
143
+ [
144
+ "qtype_score-0",
145
+ 0,
146
+ 873,
147
+ 5,
148
+ 0.0
149
+ ],
150
+ [
151
+ "multi_block_3k-0",
152
+ 0,
153
+ 2927,
154
+ 5,
155
+ 0.0
156
+ ],
157
+ [
158
+ "multi_block_3k-1",
159
+ 0,
160
+ 2932,
161
+ 2,
162
+ 0.0
163
+ ],
164
+ [
165
+ "multi_block_5k-0",
166
+ 0,
167
+ 5011,
168
+ 5,
169
+ 0.0
170
+ ],
171
+ [
172
+ "multi_block_5k-1",
173
+ 0,
174
+ 4767,
175
+ 2,
176
+ 0.0
177
+ ],
178
+ [
179
+ "multi_block_9k-s0",
180
+ 0,
181
+ 9083,
182
+ 3,
183
+ 0.0
184
+ ],
185
+ [
186
+ "multi_block_9k-s1",
187
+ 0,
188
+ 9568,
189
+ 2,
190
+ 0.0
191
+ ],
192
+ [
193
+ "prefix_boundary-2047",
194
+ 0,
195
+ 2178,
196
+ 2,
197
+ 0.0
198
+ ],
199
+ [
200
+ "prefix_boundary-2048",
201
+ 0,
202
+ 2203,
203
+ 3,
204
+ 0.0
205
+ ],
206
+ [
207
+ "prefix_boundary-2049",
208
+ 0,
209
+ 2289,
210
+ 6,
211
+ 0.0
212
+ ],
213
+ [
214
+ "prefix_boundary-4095",
215
+ 0,
216
+ 4163,
217
+ 4,
218
+ 0.0
219
+ ],
220
+ [
221
+ "prefix_boundary-4096",
222
+ 0,
223
+ 4312,
224
+ 5,
225
+ 0.0
226
+ ],
227
+ [
228
+ "prefix_boundary-4097",
229
+ 0,
230
+ 4180,
231
+ 2,
232
+ 0.0
233
+ ],
234
+ [
235
+ "prefix_boundary-6144",
236
+ 0,
237
+ 6259,
238
+ 3,
239
+ 0.0
240
+ ],
241
+ [
242
+ "catalogue_overflow-0",
243
+ 0,
244
+ 4722,
245
+ 151,
246
+ 0.0
247
+ ],
248
+ [
249
+ "catalogue_overflow-1",
250
+ 0,
251
+ 4721,
252
+ 151,
253
+ 0.0
254
+ ],
255
+ [
256
+ "catalogue_overflow-2",
257
+ 0,
258
+ 4729,
259
+ 151,
260
+ 0.0
261
+ ],
262
+ [
263
+ "catalogue_overflow_long_state-0",
264
+ 0,
265
+ 8806,
266
+ 151,
267
+ 0.0
268
+ ],
269
+ [
270
+ "many_options-0",
271
+ 0,
272
+ 500,
273
+ 75,
274
+ 0.0
275
+ ],
276
+ [
277
+ "many_options-1",
278
+ 0,
279
+ 423,
280
+ 64,
281
+ 0.0
282
+ ],
283
+ [
284
+ "long_16k-0",
285
+ 0,
286
+ 15900,
287
+ 6,
288
+ 0.0
289
+ ],
290
+ [
291
+ "long_16k-1",
292
+ 0,
293
+ 16384,
294
+ 2,
295
+ 0.0
296
+ ],
297
+ [
298
+ "long_16k-2",
299
+ 0,
300
+ 16800,
301
+ 151,
302
+ 0.0
303
+ ],
304
+ [
305
+ "long_24k-0",
306
+ 0,
307
+ 24000,
308
+ 2,
309
+ 0.0
310
+ ],
311
+ [
312
+ "long_24k-1",
313
+ 0,
314
+ 25600,
315
+ 6,
316
+ 0.0
317
+ ]
318
+ ]
319
+ }
validation/runtime/fixture_verify_bf16.json ADDED
@@ -0,0 +1,319 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "precision": "bf16",
3
+ "dev_seconds": 415.6,
4
+ "default_T": 0.8278650620942867,
5
+ "fixture_requests": 35,
6
+ "fixture_questions": 43,
7
+ "fixture_tokens": 214700,
8
+ "overflow_q": 5,
9
+ "max_abs_dscore": 0.0,
10
+ "max_abs_dprob_vs_softmax_dev_over_T": 0.0,
11
+ "text_e2e_max_abs_dscore_vs_dev": 0.0,
12
+ "text_ids_equal_dev": true,
13
+ "score_many_vs_separate_fresh_decide_max_abs": 0.0,
14
+ "order_invariance_max_abs": 0.0,
15
+ "runtime_seconds": 293.1,
16
+ "per_question": [
17
+ [
18
+ "multi_question-0",
19
+ 0,
20
+ 450,
21
+ 5,
22
+ 0.0
23
+ ],
24
+ [
25
+ "multi_question-0",
26
+ 1,
27
+ 364,
28
+ 4,
29
+ 0.0
30
+ ],
31
+ [
32
+ "multi_question-0",
33
+ 2,
34
+ 356,
35
+ 2,
36
+ 0.0
37
+ ],
38
+ [
39
+ "multi_question-0",
40
+ 3,
41
+ 345,
42
+ 4,
43
+ 0.0
44
+ ],
45
+ [
46
+ "multi_question-0",
47
+ 4,
48
+ 319,
49
+ 2,
50
+ 0.0
51
+ ],
52
+ [
53
+ "multi_question-1",
54
+ 0,
55
+ 5416,
56
+ 2,
57
+ 0.0
58
+ ],
59
+ [
60
+ "multi_question-1",
61
+ 1,
62
+ 5471,
63
+ 2,
64
+ 0.0
65
+ ],
66
+ [
67
+ "multi_question-1",
68
+ 2,
69
+ 5425,
70
+ 2,
71
+ 0.0
72
+ ],
73
+ [
74
+ "multi_question-1",
75
+ 3,
76
+ 5517,
77
+ 6,
78
+ 0.0
79
+ ],
80
+ [
81
+ "multi_question-1",
82
+ 4,
83
+ 5428,
84
+ 2,
85
+ 0.0
86
+ ],
87
+ [
88
+ "short_sb-themes-0",
89
+ 0,
90
+ 105,
91
+ 2,
92
+ 0.0
93
+ ],
94
+ [
95
+ "short_sb-public_ingests-0",
96
+ 0,
97
+ 382,
98
+ 16,
99
+ 0.0
100
+ ],
101
+ [
102
+ "short_sb-di_retrieval-0",
103
+ 0,
104
+ 197,
105
+ 2,
106
+ 0.0
107
+ ],
108
+ [
109
+ "short_sb-di_knowledge-0",
110
+ 0,
111
+ 72,
112
+ 4,
113
+ 0.0
114
+ ],
115
+ [
116
+ "short_sb-hard_short-0",
117
+ 0,
118
+ 517,
119
+ 3,
120
+ 0.0
121
+ ],
122
+ [
123
+ "short_sb-format-0",
124
+ 0,
125
+ 166,
126
+ 21,
127
+ 0.0
128
+ ],
129
+ [
130
+ "qtype_choice-0",
131
+ 0,
132
+ 594,
133
+ 3,
134
+ 0.0
135
+ ],
136
+ [
137
+ "qtype_noul-0",
138
+ 0,
139
+ 246,
140
+ 2,
141
+ 0.0
142
+ ],
143
+ [
144
+ "qtype_score-0",
145
+ 0,
146
+ 873,
147
+ 5,
148
+ 0.0
149
+ ],
150
+ [
151
+ "multi_block_3k-0",
152
+ 0,
153
+ 2927,
154
+ 5,
155
+ 0.0
156
+ ],
157
+ [
158
+ "multi_block_3k-1",
159
+ 0,
160
+ 2932,
161
+ 2,
162
+ 0.0
163
+ ],
164
+ [
165
+ "multi_block_5k-0",
166
+ 0,
167
+ 5011,
168
+ 5,
169
+ 0.0
170
+ ],
171
+ [
172
+ "multi_block_5k-1",
173
+ 0,
174
+ 4767,
175
+ 2,
176
+ 0.0
177
+ ],
178
+ [
179
+ "multi_block_9k-s0",
180
+ 0,
181
+ 9083,
182
+ 3,
183
+ 0.0
184
+ ],
185
+ [
186
+ "multi_block_9k-s1",
187
+ 0,
188
+ 9568,
189
+ 2,
190
+ 0.0
191
+ ],
192
+ [
193
+ "prefix_boundary-2047",
194
+ 0,
195
+ 2178,
196
+ 2,
197
+ 0.0
198
+ ],
199
+ [
200
+ "prefix_boundary-2048",
201
+ 0,
202
+ 2203,
203
+ 3,
204
+ 0.0
205
+ ],
206
+ [
207
+ "prefix_boundary-2049",
208
+ 0,
209
+ 2289,
210
+ 6,
211
+ 0.0
212
+ ],
213
+ [
214
+ "prefix_boundary-4095",
215
+ 0,
216
+ 4163,
217
+ 4,
218
+ 0.0
219
+ ],
220
+ [
221
+ "prefix_boundary-4096",
222
+ 0,
223
+ 4312,
224
+ 5,
225
+ 0.0
226
+ ],
227
+ [
228
+ "prefix_boundary-4097",
229
+ 0,
230
+ 4180,
231
+ 2,
232
+ 0.0
233
+ ],
234
+ [
235
+ "prefix_boundary-6144",
236
+ 0,
237
+ 6259,
238
+ 3,
239
+ 0.0
240
+ ],
241
+ [
242
+ "catalogue_overflow-0",
243
+ 0,
244
+ 4722,
245
+ 151,
246
+ 0.0
247
+ ],
248
+ [
249
+ "catalogue_overflow-1",
250
+ 0,
251
+ 4721,
252
+ 151,
253
+ 0.0
254
+ ],
255
+ [
256
+ "catalogue_overflow-2",
257
+ 0,
258
+ 4729,
259
+ 151,
260
+ 0.0
261
+ ],
262
+ [
263
+ "catalogue_overflow_long_state-0",
264
+ 0,
265
+ 8806,
266
+ 151,
267
+ 0.0
268
+ ],
269
+ [
270
+ "many_options-0",
271
+ 0,
272
+ 500,
273
+ 75,
274
+ 0.0
275
+ ],
276
+ [
277
+ "many_options-1",
278
+ 0,
279
+ 423,
280
+ 64,
281
+ 0.0
282
+ ],
283
+ [
284
+ "long_16k-0",
285
+ 0,
286
+ 15900,
287
+ 6,
288
+ 0.0
289
+ ],
290
+ [
291
+ "long_16k-1",
292
+ 0,
293
+ 16384,
294
+ 2,
295
+ 0.0
296
+ ],
297
+ [
298
+ "long_16k-2",
299
+ 0,
300
+ 16800,
301
+ 151,
302
+ 0.0
303
+ ],
304
+ [
305
+ "long_24k-0",
306
+ 0,
307
+ 24000,
308
+ 2,
309
+ 0.0
310
+ ],
311
+ [
312
+ "long_24k-1",
313
+ 0,
314
+ 25600,
315
+ 6,
316
+ 0.0
317
+ ]
318
+ ]
319
+ }
validation/runtime/render_verify.json ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "requests": 463,
3
+ "questions": 756,
4
+ "ok_identical": 743,
5
+ "budget_both": 6,
6
+ "error_both": 6,
7
+ "mismatch": [
8
+ {
9
+ "q_repr": "{'t': 'score', 'ins': '\\n', 'crit': [' <tool_response>', '\\n<tool_response>->', '<|box_start|>\\nOptions:\\nOption 2: ', '<tool_response> -> yesabc', '<|endoftext|><|endoftext|>Option 1 ->', '', '<|video_pad|>\ud83d\ude42Option 2: ', '<tool_call>Judge each numbered option in the complete catalogue above:\\n<think",
10
+ "state_repr": "'abc<|im_start|><|vision_pad|>Judge each numbered option in the complete catalogue above:\\n<think>\\n",
11
+ "id": "special:16",
12
+ "q": 0,
13
+ "ref": [
14
+ "ok",
15
+ 245
16
+ ],
17
+ "rt": "error",
18
+ "ref_err": null,
19
+ "rt_err": [
20
+ "QuestionError"
21
+ ]
22
+ }
23
+ ],
24
+ "overflow_q": 17,
25
+ "max_tokens": 25600,
26
+ "max_k": 1000,
27
+ "prefix_path_mismatch": 0,
28
+ "boundary_head_lens": [
29
+ 2046,
30
+ 2047,
31
+ 2048,
32
+ 2095,
33
+ 2096
34
+ ],
35
+ "state_prefix_lens": [
36
+ 2047,
37
+ 2048,
38
+ 2049,
39
+ 4095,
40
+ 4096,
41
+ 4097,
42
+ 20479,
43
+ 20480,
44
+ 20481
45
+ ],
46
+ "overflow_budget_edge_total": 25600,
47
+ "budget_ids": [
48
+ "total_edge:18657",
49
+ "total_edge:18658",
50
+ "bigk:1500",
51
+ "maxlen:64",
52
+ "maxlen:300",
53
+ "huge_question"
54
+ ],
55
+ "error_ids": [
56
+ [
57
+ "ctrl:1",
58
+ "TypeError",
59
+ "TypeError"
60
+ ],
61
+ [
62
+ "score_11",
63
+ "RecordError",
64
+ "QuestionError"
65
+ ],
66
+ [
67
+ "score_1",
68
+ "RecordError",
69
+ "QuestionError"
70
+ ],
71
+ [
72
+ "choice_empty",
73
+ "RecordError",
74
+ "QuestionError"
75
+ ],
76
+ [
77
+ "choice_numkeys",
78
+ "ValueError",
79
+ "TypeError"
80
+ ],
81
+ [
82
+ "bad_type",
83
+ "RecordError",
84
+ "QuestionError"
85
+ ]
86
+ ],
87
+ "n_mismatch": 1
88
+ }
validation/runtime/reverify_mlx.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "runtime_file": "jev_style_decision_mlx.py",
3
+ "runtime_sha256": "025daab9b374d740e743c519bd3aa3385791e08c81f4515b2ae525af57788030",
4
+ "shared_core_sha256": "56deae095206fc59b9a30d8626ac5d27cacddcd94d7c7f08f6f4bc53d326a002",
5
+ "what": "MLX bf16 (loaded from the repository root, precision bf16), runtime defaults",
6
+ "loaded_from": "runs/macjev/hf_staging/Jev-Style-2B-Decision-v3-MLX",
7
+ "verify_manifest": true,
8
+ "compared_with": {
9
+ "file": "runs/macjev/release_2b/parity/pred_mlx_bf16_base.jsonl",
10
+ "sha256": "c42232870ab8c5410b4f45dc85729249d467b9c0bd5d24213d085e4479a3a60e"
11
+ },
12
+ "fixture": {
13
+ "file": "runs/macjev/runtime_v2_dev/reference_base/fixture.jsonl",
14
+ "sha256": "2b16d25eafa70c00a1106ba51b94890821b51ecb4fa393abb12403c8739ef314"
15
+ },
16
+ "questions": 43,
17
+ "skipped_questions": 0,
18
+ "max_abs_score_diff": 0.0,
19
+ "top1_same": "43/43",
20
+ "token_or_overflow_mismatches": [],
21
+ "seconds": 137.2,
22
+ "finished_unix": 1790449492.3913028
23
+ }