diff --git a/.gitattributes b/.gitattributes index a6344aac8c09253b3b630fb776ae94478aa0275b..3a60c8e8d37e742f8419cf2789f61a94183e03cb 100644 --- a/.gitattributes +++ b/.gitattributes @@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zip filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text +4B-5949/tokenizer.json filter=lfs diff=lfs merge=lfs -text +9B-3000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +9B-5949/tokenizer.json filter=lfs diff=lfs merge=lfs -text +TECHNICAL_REPORT.pdf filter=lfs diff=lfs merge=lfs -text diff --git a/4B-5949/LICENSE b/4B-5949/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/4B-5949/LICENSE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/4B-5949/README.md b/4B-5949/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7428bab3cb20cc0a7f6ba26e69bf76c9ff192be5 --- /dev/null +++ b/4B-5949/README.md @@ -0,0 +1,33 @@ +# APUS-OpenJev-v1 · 4B + +A decision model for browser agents and business workflows. This directory contains standalone BF16 weights and a runtime with selectable `effort="low"` and `effort="high"` compute budgets. + +[Model family](../README.md) · [Architecture](../ARCHITECTURE.md) · [Runtime guide](RUNTIME.md) + +## Quick start + +Use a CUDA-capable PyTorch environment. + +```bash +python -m pip install huggingface_hub +hf auth login +hf download apus-ailab/APUS-OpenJev-v1 \ + --include "4B-5949/*" --local-dir ./APUS-OpenJev-v1 +cd ./APUS-OpenJev-v1/4B-5949 +python -m pip install -r requirements.txt +python examples.py . --device cuda:0 --effort high +``` + +The included runtime provides compute-budget selection. Use `high` for text generation. + +## Evaluation + +With the full compute budget, this merged model scores **66/80 (82.50%)** on the [Frozen80 development panel](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921): browser action selection, principle-based judgment, evidence-based questions, natural language inference, and attribute decisions. + +This reused development panel is an engineering reference, not an independent benchmark. BF16 merging changes some candidate probabilities; decision thresholds require revalidation. See [evaluation results](merged-evaluation.json) and [runtime checks](evaluation/runtime-smoke.json) for details. + +## Provenance + +We thank the Qwen team for the [Qwen3.5-4B](https://huggingface.co/Qwen/Qwen3.5-4B) base model. Training details and source records are in [training.md](training.md); artifact hashes are in [release-manifest.json](release-manifest.json). Consult [LICENSE](LICENSE) and the [base-model](provenance/BASE-LICENSE.txt) and [source-project](provenance/SOURCE-PROJECT-LICENSE.txt) notices. + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab) diff --git a/4B-5949/RUNTIME.md b/4B-5949/RUNTIME.md new file mode 100644 index 0000000000000000000000000000000000000000..e6e76c9d13b52c2f3a872136072081aacdd05404 --- /dev/null +++ b/4B-5949/RUNTIME.md @@ -0,0 +1,55 @@ +# xDAN-openJet merged reference runtime + +本目录是可随 HF merged 仓库发布的完整 Python 源码闭包,不需要安装 ms-swift、PEFT 或原项目。发布选择为 **4B checkpoint-5949**、**9B checkpoint-3000** 与 **9B checkpoint-5949**;各目录必须保留自己的 `depth_config.json`、完整 Qwen config、tokenizer、chat template、generation config 和 merged safetensors。不得将不同 checkpoint 的权重或 shallow/full 结果拼在一起。 + +## 运行 + +使用 CUDA 对应 PyTorch 2.8.0 构建,安装 `requirements.txt`。运行依赖 Transformer 私有模型层接口,所以严格要求 `transformers==5.16.1`;版本升级需重做层级及数值验证。这是原生 PyTorch 单卡单请求参考实现,不是 vLLM 服务。 + +先把发布仓库的**固定 commit**完整下载到本机目录。仓库根目录有本目录中的 `openjet_runtime/` 与 `examples.py` 时: + +```bash +python -m pip install -r requirements.txt +python examples.py ./model-snapshot --device cuda:0 --effort both +python examples.py ./model-snapshot --device cuda:0 --effort high --text +``` + +若代码与权重同在下载快照根目录,进入快照后把 `./model-snapshot` 改为 `.`。 + +```python +from openjet_runtime import OpenJet +from examples import decision_examples + +model = OpenJet.from_pretrained("./model-snapshot") +two_candidates, sixteen_candidates = decision_examples() +print(model.decide(two_candidates, effort="low")) +print(model.decide(sixteen_candidates, effort="high")) +print(model.generate_text("Return only the text: red shoes", effort="high", max_new_tokens=32)) +``` + +`examples.py` 中两候选工作流、16 候选浏览器和 TYPE 是接口演示,不是声称模型已通过的 benchmark。需要针对业务设计 prompt 和验证答案。 + +## 接口和执行语义 + +- `decide(request, effort)` 输入字段与原 `jev.dynamic.prompt.v2` 相同:`id/group_id/state/instructions/primitive/criteria`;每个候选有非空 `id` 和 `description`,2–16 个,ID 不重复。标签为 A–P,编译器验证每个标签在真实回答边界恰为一个 token。没有 gold 输入需求。 +- `primitive="choice"` 返回 `choice` 与按输入顺序映射的 `probabilities`;`noul/score_level` 使用 `contracts.py` 中固定 Yes/No 候选,返回 `yes_probability`。`score_level` 是单个命题的判断,不能当作完整序数 Score API。 +- `effort="low"` 执行 `depth_config.exit_depth`(这两项发布预期16),共享原 LM final norm 与候选行投影;`high` 执行 `full_depth`(预期32),保持原评测中的标准模型前向+完整 LM head 路径。读取 config,不凭参数规模推测层数。 +- 所有输入采用 tokenizer 自带 chat template、`enable_thinking=False`,超过8192 token直接报错。不会默默截断 state、instructions 或候选。 +- `probabilities` 是在当前候选集上的相对 softmax,**未经概率校准**;合并不自动带来可信置信度或校准保证。 +- `generate_text` 为 TYPE 文本保留的贪心参考路径;每个 token 重新计算前缀,方便与原评测逐 token 核对,但不适合宣传 tokens/s。达到上限明确返回 `finish_reason="length"`。 +- `both` 示例分别调用两个 effort;没有自动路由、不承诺共享两次调用的前缀缓存。本次便携发布不包含 KV 广播引擎、vLLM 插件、TypeSafe HTTP server 或多模态输入能力。 + +## 合并验收(GPU,不能用静态测试代替) + +1. 固定同一 base revision、adapter SHA、tokenizer、chat template、dtype、attention backend 和 Transformers 版本,记录 merge 前后权重身份。保留 adapter 原文件。 +2. 同进程/新进程分别加载 base+adapter 和 merged;比较固定2候选、16候选、长输入、80题面板两 effort 的 token IDs、候选排序、logits、probabilities 和最终ID。保存逐题差异与最大绝对差,不能只比 aggregate accuracy。 +3. BF16 merge 会舍入:不能预先声明 bitwise一致或把漂移简单解释为无害;应报告数值误差和所有预测翻转。若超过事先制定的容忍值,停止发布数值等价结论,考虑 FP32 merge/存储再独立评测。 +4. fresh reload 验证 `depth_config` 与实际层数、模块边界一致。用层 forward hooks 检查 low只执行浅层、high执行全层;hooks会触发候选头保守fallback,不拿该测量做性能报告。 +5. TYPE短样本比较 token序列和EOS;分别测试空输入、非法候选/重复ID、超长输入明确失败。 +6. 在干净环境固定 HF revision 下载,运行本目录例子和同一小面板。记录显存、依赖、GPU型号以及权重checksum。成功加载只是第一关,不能当作质量或吞吐验收。 + +原始数值等价门结果:`False`,保留原结果;决策完全一致门:`True`,一致 `160/160`。独立运行时 GPU 验收状态:`passed`。若数值门失败,此 BF16 包作为独立重评版本,禁止直接迁移概率/拒答/路由阈值;见 `merged-evaluation.json`。 + +## 源码来历 + +`contracts.py`、`candidate_projection.py` 与 `early_exit.py` 从已有本地实现原样提取(最后一项仅调整相对 import);`source-provenance.json` 记录源路径与两端 SHA256。`runtime.py` 是最小加载及接口层;high 与 TYPE 分别对应原 `HFDecisionEngine._forward_batch` 和 `package_eval.prefix_next_token` 的执行语义。采用现有模型类:`qwen3_5` → `Qwen3_5ForConditionalGeneration`;`qwen3_5_text` → `AutoModelForCausalLM`。没有新增学习参数。 diff --git a/4B-5949/chat_template.jinja b/4B-5949/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..a585dec894e63da457d9440ec6aa7caa16d20860 --- /dev/null +++ b/4B-5949/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/4B-5949/config.json b/4B-5949/config.json new file mode 100644 index 0000000000000000000000000000000000000000..46fa47942518dbbe94bd4ef4f345f454663743e4 --- /dev/null +++ b/4B-5949/config.json @@ -0,0 +1,109 @@ +{ + "architectures": [ + "Qwen3_5ForConditionalGeneration" + ], + "dtype": "bfloat16", + "image_token_id": 248056, + "model_type": "qwen3_5", + "text_config": { + "attention_bias": false, + "attention_dropout": 0.0, + "attn_output_gate": true, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 248044, + "full_attention_interval": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9216, + "layer_types": [ + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention" + ], + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 16, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "mamba_ssm_dtype": "float32", + "max_position_embeddings": 262144, + "mlp_only_layers": [], + "model_type": "qwen3_5_text", + "mtp_num_hidden_layers": 1, + "mtp_use_dedicated_embeddings": false, + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 4, + "pad_token_id": null, + "partial_rotary_factor": 0.25, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "mrope_interleaved": true, + "mrope_section": [ + 11, + 11, + 10 + ], + "partial_rotary_factor": 0.25, + "rope_theta": 10000000, + "rope_type": "default" + }, + "tie_word_embeddings": true, + "use_cache": true, + "vocab_size": 248320 + }, + "tie_word_embeddings": true, + "transformers_version": "5.16.1", + "video_token_id": 248057, + "vision_config": { + "deepstack_visual_indexes": [], + "depth": 24, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1024, + "in_channels": 3, + "initializer_range": 0.02, + "intermediate_size": 4096, + "model_type": "qwen3_5_vision", + "num_heads": 16, + "num_position_embeddings": 2304, + "out_hidden_size": 2560, + "patch_size": 16, + "spatial_merge_size": 2, + "temporal_patch_size": 2 + }, + "vision_end_token_id": 248054, + "vision_start_token_id": 248053 +} diff --git a/4B-5949/decision-release.json b/4B-5949/decision-release.json new file mode 100644 index 0000000000000000000000000000000000000000..ef6463b39b909061f60475b8a92b3111f1234ef9 --- /dev/null +++ b/4B-5949/decision-release.json @@ -0,0 +1,35 @@ +{ + "passed": true, + "scope": "Independent BF16 merged variant, frozen80 decision parity; NOT a probability-equivalent replacement", + "post_observation_scope_amendment": true, + "original_numerical_gate_passed": false, + "original_gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "original_comparison_sha256": "97cc94ddebb4f321d3d4ed29c89a0d8b1b3d6570e48bff427c635ba82f67862e", + "decision_agreement": 160, + "total_decisions": 160, + "independent_questions": 80, + "probability_calibration_transfer_validated": false, + "automatic_routing_validated": false, + "required_followup": "Recalibrate all probability, rejection and routing thresholds; independently evaluate new tasks", + "depths": { + "16": { + "before_correct": 61, + "after_correct": 61, + "changed_decisions": [], + "max_probability_abs_difference": 0.09226870536804199, + "mean_probability_abs_difference": 0.001908167676742778, + "max_logit_abs_difference": 0.5078125, + "mean_logit_abs_difference": 0.026137218475341797 + }, + "32": { + "before_correct": 66, + "after_correct": 66, + "changed_decisions": [], + "max_probability_abs_difference": 0.06145721673965454, + "mean_probability_abs_difference": 0.0017077110185891797, + "max_logit_abs_difference": 0.875, + "mean_logit_abs_difference": 0.03171875 + } + }, + "publication_requires": "Further portable-runtime160, weight arithmetic, full file hashes and fresh HF reload gates" +} diff --git a/4B-5949/depth_config.json b/4B-5949/depth_config.json new file mode 100644 index 0000000000000000000000000000000000000000..78d0ebd21a4c606e7baf828ab18c6082bf90efdb --- /dev/null +++ b/4B-5949/depth_config.json @@ -0,0 +1,10 @@ +{ + "prompt_version": "jev.dynamic.prompt.v2", + "exit_depth": 16, + "full_depth": 32, + "model_series": "xDAN-openJet", + "checkpoint_step": 5949, + "source_depth_config_sha256": "6856cf257aa6eeb24dda20702ace04696b5d3db19d647705dedd00cca6f756e6", + "training_mode": "two_exit", + "automatic_routing_validated": false +} diff --git a/4B-5949/evaluation/runtime-smoke.json b/4B-5949/evaluation/runtime-smoke.json new file mode 100644 index 0000000000000000000000000000000000000000..7ab66949a38b9184e20813b5e162d0e55bed1f8f --- /dev/null +++ b/4B-5949/evaluation/runtime-smoke.json @@ -0,0 +1,187 @@ +{ + "status": "passed", + "runtime_sha256": { + "openjet_runtime/__init__.py": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944", + "openjet_runtime/runtime.py": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c", + "openjet_runtime/early_exit.py": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566", + "openjet_runtime/contracts.py": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "openjet_runtime/candidate_projection.py": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "synthetic_examples": [ + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.9961305856704712, + "refund": 0.0038693908136337996 + }, + "choice": "close", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 103, + "logits": [ + 5.09375, + -0.45703125 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.9996199607849121, + "refund": 0.0003799845289904624 + }, + "choice": "close", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 103, + "logits": [ + 20.375, + 12.5 + ], + "projection": "full_head", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 0.06669197976589203, + "click-2": 0.016471756622195244, + "click-3": 0.03487071022391319, + "click-4": 0.06072373315691948, + "click-5": 0.025511953979730606, + "click-6": 0.03125790134072304, + "click-7": 0.0377042256295681, + "click-8": 0.05234713479876518, + "click-9": 0.07498379796743393, + "click-10": 0.027584997937083244, + "click-11": 0.04879289120435715, + "click-12": 0.042062100023031235, + "click-13": 0.14767390489578247, + "click-14": 0.10910079628229141, + "click-15": 0.08630583435297012, + "click-16": 0.13791632652282715 + }, + "choice": "click-13", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 383, + "logits": [ + -1.1796875, + -2.578125, + -1.828125, + -1.2734375, + -2.140625, + -1.9375, + -1.75, + -1.421875, + -1.0625, + -2.0625, + -1.4921875, + -1.640625, + -0.384765625, + -0.6875, + -0.921875, + -0.453125 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 0.00045352030429057777, + "click-2": 0.00040023025940172374, + "click-3": 0.00017760110495146364, + "click-4": 0.0002142278099199757, + "click-5": 0.0002584080502856523, + "click-6": 0.00024275195028167218, + "click-7": 0.0003532019618432969, + "click-8": 0.0004827698867302388, + "click-9": 0.0013123045209795237, + "click-10": 0.0004827698867302388, + "click-11": 0.0035672136582434177, + "click-12": 0.9890894293785095, + "click-13": 0.0012327961158007383, + "click-14": 0.0007959529175423086, + "click-15": 0.0001890553830889985, + "click-16": 0.0007477285689674318 + }, + "choice": "click-12", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 383, + "logits": [ + 12.5625, + 12.4375, + 11.625, + 11.8125, + 12.0, + 11.9375, + 12.3125, + 12.625, + 13.625, + 12.625, + 14.625, + 20.25, + 13.5625, + 13.125, + 11.6875, + 13.0625 + ], + "projection": "full_head", + "calibrated": false + } + ], + "text_smoke": { + "low": { + "text": "\u4f60\u597d / / / / ...etc\u4e4e--SsN [sS", + "token_ids": [ + 109266, + 593, + 593, + 593, + 593, + 2423, + 11763, + 97091, + 12, + 12, + 50, + 82, + 45, + 498, + 82, + 50 + ], + "effort": "low", + "executed_layers_per_token": 16, + "finish_reason": "length", + "prompt_tokens": 28 + }, + "high": { + "text": "red shoes\n", + "token_ids": [ + 1114, + 14850, + 248046, + 198, + 248044 + ], + "effort": "high", + "executed_layers_per_token": 32, + "finish_reason": "eos", + "prompt_tokens": 28 + } + }, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation" +} diff --git a/4B-5949/evaluation/weight-arithmetic.json b/4B-5949/evaluation/weight-arithmetic.json new file mode 100644 index 0000000000000000000000000000000000000000..ca789a0f99caa51c2282986d9bedccda3d93d7b0 --- /dev/null +++ b/4B-5949/evaluation/weight-arithmetic.json @@ -0,0 +1,908 @@ +{ + "status": "passed", + "base_mtp_keys_not_loaded_by_transformers": [ + "mtp.fc.weight", + "mtp.layers.0.input_layernorm.weight", + "mtp.layers.0.mlp.down_proj.weight", + "mtp.layers.0.mlp.gate_proj.weight", + "mtp.layers.0.mlp.up_proj.weight", + "mtp.layers.0.post_attention_layernorm.weight", + "mtp.layers.0.self_attn.k_norm.weight", + "mtp.layers.0.self_attn.k_proj.weight", + "mtp.layers.0.self_attn.o_proj.weight", + "mtp.layers.0.self_attn.q_norm.weight", + "mtp.layers.0.self_attn.q_proj.weight", + "mtp.layers.0.self_attn.v_proj.weight", + "mtp.norm.weight", + "mtp.pre_fc_norm_embedding.weight", + "mtp.pre_fc_norm_hidden.weight" + ], + "size": "4B", + "targeted_weight_matrices": 176, + "adapter_tensors": 352, + "all_targeted_weights_exact": true, + "scope": "FP32 LoRA arithmetic followed by BF16 rounding; not forward/probability equivalence", + "adapter_sha256": "71154f60ec72c55d2c6c6147b9cdda9cc9a52d46297f3074dded7cdc9f5bc344", + "checks": [ + { + "weight": "model.language_model.layers.0.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + } + ] +} diff --git a/4B-5949/examples.py b/4B-5949/examples.py new file mode 100644 index 0000000000000000000000000000000000000000..c942c70fb5ee82412ae703b04eea3ea58d4226b8 --- /dev/null +++ b/4B-5949/examples.py @@ -0,0 +1,61 @@ +"""Run against a local merged HF snapshot; no project-local dependencies.""" + +import argparse +import json + +from openjet_runtime import OpenJet + + +def decision_examples(): + binary = { + "id": "example-binary", + "group_id": "example-binary", + "primitive": "choice", + "state": "Order 731 has been delivered. The customer's message says thank you.", + "instructions": "Select the appropriate next workflow action.", + "criteria": [ + {"id": "close", "description": "Close the resolved support ticket."}, + {"id": "refund", "description": "Refund an undelivered order."}, + ], + } + browser = { + "id": "example-browser", + "group_id": "example-browser", + "primitive": "choice", + "state": "A settings page has 16 visible buttons labeled Page 1 through Page 16.", + "instructions": "Navigate to Page 12 by choosing its matching button.", + "criteria": [ + {"id": f"click-{i}", "description": f"Click the Page {i} button."} + for i in range(1, 17) + ], + } + return binary, browser + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("model", help="Local merged snapshot directory") + parser.add_argument("--device", default="cuda:0") + parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") + parser.add_argument("--effort", choices=["low", "high", "both"], default="both") + parser.add_argument( + "--text", action="store_true", help="Also run slow TYPE reference" + ) + args = parser.parse_args() + runtime = OpenJet.from_pretrained(args.model, args.device, args.dtype) + efforts = ("low", "high") if args.effort == "both" else (args.effort,) + for effort in efforts: + for request in decision_examples(): + result = runtime.decide(request, effort) + print(json.dumps({"example": request["id"], **result}, ensure_ascii=False)) + if args.text: + result = runtime.generate_text( + "Return only the literal text to type into a search box for 'red shoes'.", + effort=effort, + max_new_tokens=32, + ) + print(json.dumps({"example": "browser-type", **result}, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/4B-5949/generation_config.json b/4B-5949/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..f0b25ab7f068b3916a0b4e942812ee859e14c813 --- /dev/null +++ b/4B-5949/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "eos_token_id": 248044, + "transformers_version": "5.16.1", + "use_cache": true +} diff --git a/4B-5949/merge-provenance.json b/4B-5949/merge-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..2b8b7e49e1ae9352b80f5975d4011854f9c31e28 --- /dev/null +++ b/4B-5949/merge-provenance.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-4B", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "checkpoint_step": 5949, + "adapter_sha256": "71154f60ec72c55d2c6c6147b9cdda9cc9a52d46297f3074dded7cdc9f5bc344", + "adapter_config_sha256": "c6caee3818ca1f6c8539e47fac9b9fa818d28b61a4bbb05f6f8444c0dd639e45", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/4B-5949/merged-evaluation.json b/4B-5949/merged-evaluation.json new file mode 100644 index 0000000000000000000000000000000000000000..fd7107d0ca7654707ff3ea64dc86dd2a4ba75ff8 --- /dev/null +++ b/4B-5949/merged-evaluation.json @@ -0,0 +1,83 @@ +{ + "scope": "80 fixed requests, both explicit efforts; merged-model comparison, not independent generalization evidence", + "checkpoint_step": 5949, + "base_id": "Qwen/Qwen3.5-4B", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "comparison": { + "passed": false, + "gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "depths": { + "16": { + "before_correct": 61, + "after_correct": 61, + "changed_decisions": [], + "max_probability_abs_difference": 0.09226870536804199, + "mean_probability_abs_difference": 0.001908167676742778, + "max_logit_abs_difference": 0.5078125, + "mean_logit_abs_difference": 0.026137218475341797 + }, + "32": { + "before_correct": 66, + "after_correct": 66, + "changed_decisions": [], + "max_probability_abs_difference": 0.06145721673965454, + "mean_probability_abs_difference": 0.0017077110185891797, + "max_logit_abs_difference": 0.875, + "mean_logit_abs_difference": 0.03171875 + } + }, + "before_sha256": "7be98e54b67be84ae97a4152adeddd47a351e224cbe9345bae8ca8f44c265964", + "after_sha256": "ac22ede19c6ade99388038292644df50e9891dff162acdb10029bb3bd343537e", + "verified_at_unix": 1789983995.6865954 + }, + "runtime_validation": { + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "60a80902675971ff6a054377852b936eec90f49f466eb7b8b077eee7b979f230" + }, + "decision_release": { + "passed": true, + "scope": "Independent BF16 merged variant, frozen80 decision parity; NOT a probability-equivalent replacement", + "post_observation_scope_amendment": true, + "original_numerical_gate_passed": false, + "original_gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "original_comparison_sha256": "97cc94ddebb4f321d3d4ed29c89a0d8b1b3d6570e48bff427c635ba82f67862e", + "decision_agreement": 160, + "total_decisions": 160, + "independent_questions": 80, + "probability_calibration_transfer_validated": false, + "automatic_routing_validated": false, + "required_followup": "Recalibrate all probability, rejection and routing thresholds; independently evaluate new tasks", + "depths": { + "16": { + "before_correct": 61, + "after_correct": 61, + "changed_decisions": [], + "max_probability_abs_difference": 0.09226870536804199, + "mean_probability_abs_difference": 0.001908167676742778, + "max_logit_abs_difference": 0.5078125, + "mean_logit_abs_difference": 0.026137218475341797 + }, + "32": { + "before_correct": 66, + "after_correct": 66, + "changed_decisions": [], + "max_probability_abs_difference": 0.06145721673965454, + "mean_probability_abs_difference": 0.0017077110185891797, + "max_logit_abs_difference": 0.875, + "mean_logit_abs_difference": 0.03171875 + } + }, + "publication_requires": "Further portable-runtime160, weight arithmetic, full file hashes and fresh HF reload gates" + }, + "reviewed_variant": {}, + "decision_parity_passed": true, + "decision_agreement": 160, + "weight_arithmetic_status": "passed", + "raw_inputs_included": false +} diff --git a/4B-5949/model-00001-of-00003.safetensors b/4B-5949/model-00001-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e1cb2ff5615c5f1ee25879e5aec0fc581246ae10 --- /dev/null +++ b/4B-5949/model-00001-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c04bef62040612a3376c144014c194687cdc19b18c3a807ee1136b4a713cbb0a +size 3991298872 diff --git a/4B-5949/model-00002-of-00003.safetensors b/4B-5949/model-00002-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..05dcf4b8c1f4aaa8f7ac099b0d15b625c947f831 --- /dev/null +++ b/4B-5949/model-00002-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7891d5b76b686893680e9cc074c2e17a788ff0cb03f64cc2ae4150bd804299a2 +size 3979833152 diff --git a/4B-5949/model-00003-of-00003.safetensors b/4B-5949/model-00003-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..27d854e32e8a11f7786b9b9bd0d61de983bb72e6 --- /dev/null +++ b/4B-5949/model-00003-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a88eecd668f83cad773bd67c9c7e6e466c1746d489c55d906b48398a6679db65 +size 1107487880 diff --git a/4B-5949/model.safetensors.index.json b/4B-5949/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..c2e50d3f685604123a3d207c5d0b65b36ec55339 --- /dev/null +++ b/4B-5949/model.safetensors.index.json @@ -0,0 +1,731 @@ +{ + "metadata": { + "total_parameters": 4539265536, + "total_size": 9078531072 + }, + "weight_map": { + "model.language_model.embed_tokens.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.k_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.k_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.o_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.q_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.q_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.11.self_attn.v_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.2.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.20.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.3.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.k_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.k_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.o_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.q_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.q_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.v_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.30.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_qkv.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_z.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.norm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.out_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.mlp.down_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.mlp.gate_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.mlp.up_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.30.post_attention_layernorm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.input_layernorm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.mlp.down_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.mlp.gate_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.mlp.up_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.post_attention_layernorm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.k_norm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.k_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.o_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.q_norm.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.q_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.31.self_attn.v_proj.weight": "model-00003-of-00003.safetensors", + "model.language_model.layers.4.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.k_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.k_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.o_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.q_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.q_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.v_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.norm.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.0.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.1.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.4.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.5.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.6.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.7.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.8.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.9.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.merger.norm.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.norm.weight": "model-00003-of-00003.safetensors", + "model.visual.patch_embed.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.patch_embed.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.pos_embed.weight": "model-00003-of-00003.safetensors" + } +} diff --git a/4B-5949/openjet_runtime/__init__.py b/4B-5949/openjet_runtime/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..b314e3fca191cba379bb38ac6156c1f57a93ba93 --- /dev/null +++ b/4B-5949/openjet_runtime/__init__.py @@ -0,0 +1,3 @@ +from .runtime import OpenJet + +__all__ = ["OpenJet"] diff --git a/4B-5949/openjet_runtime/candidate_projection.py b/4B-5949/openjet_runtime/candidate_projection.py new file mode 100644 index 0000000000000000000000000000000000000000..7241c2a9985c91f726c2a145a52a09a90a3e6449 --- /dev/null +++ b/4B-5949/openjet_runtime/candidate_projection.py @@ -0,0 +1,108 @@ +"""Project selected native dense LM-head rows before matmul, preserving autograd. + +The caller must pass the actual output head, never an unwrapped adapter base_layer. +Unsupported heads raise; there is deliberately no automatic full-head fallback. +""" + +import torch +from torch import nn +from torch.nn import functional as F +from torch.nn.modules import module as module_hooks + + +class UnsupportedCandidateHead(ValueError): + """The head's semantics cannot be reproduced by plain selected-row linear.""" + + +def _check_head(head): + if type(head) is not nn.Linear: + raise UnsupportedCandidateHead("only exact torch.nn.Linear is supported") + if ( + head.forward.__func__ is not nn.Linear.forward + if hasattr(head.forward, "__func__") + else True + ): + raise UnsupportedCandidateHead("overridden forward is unsupported") + if head._modules or head._buffers or set(head._parameters) != {"weight", "bias"}: + raise UnsupportedCandidateHead( + "head contains extra modules, buffers or parameters" + ) + for name in ( + "_forward_hooks", + "_forward_pre_hooks", + "_backward_hooks", + "_backward_pre_hooks", + ): + if getattr(head, name, None) or getattr(module_hooks, "_global" + name, None): + raise UnsupportedCandidateHead("module hooks would be bypassed") + for value in (head.weight, head.bias): + if value is None: + continue + if ( + type(value) is not nn.Parameter + or value.is_quantized + or value.layout != torch.strided + or not value.is_floating_point() + or value.device.type == "meta" + ): + raise UnsupportedCandidateHead( + "requires ordinary dense floating-point Parameters" + ) + if head.weight is None or head.weight.shape != ( + head.out_features, + head.in_features, + ): + raise UnsupportedCandidateHead("invalid dense weight shape") + if head.bias is not None and ( + head.bias.shape != (head.out_features,) + or head.bias.dtype != head.weight.dtype + or head.bias.device != head.weight.device + ): + raise UnsupportedCandidateHead( + "bias shape, dtype or device does not match weight" + ) + + +def candidate_logits(head, hidden_states, token_ids): + """Return [..., K] logits in supplied token order, including repeated IDs. + + Dtype/autocast follow F.linear; no explicit precision conversion or detach. + Shape-dependent floating GEMM rounding may differ from full-vocabulary GEMM. + This validates indices, not tokenization, semantic labels, or probability mass. + """ + _check_head(head) + if type(hidden_states) not in (torch.Tensor, nn.Parameter): + raise TypeError("hidden_states must be an ordinary Tensor") + if ( + hidden_states.ndim < 1 + or hidden_states.shape[-1] != head.in_features + or not hidden_states.is_floating_point() + or hidden_states.layout != torch.strided + or hidden_states.device != head.weight.device + ): + raise ValueError( + "hidden_states shape, floating layout or device does not match head" + ) + if type(token_ids) is torch.Tensor: + if ( + token_ids.ndim != 1 + or token_ids.dtype != torch.long + or token_ids.device.type == "meta" + ): + raise ValueError("token_ids must be a one-dimensional int64 tensor") + indices = token_ids.to(device=head.weight.device) + elif isinstance(token_ids, (list, tuple)): + if any(type(value) is not int for value in token_ids): + raise TypeError("token IDs must be integers, not booleans or floats") + if any(value < 0 or value >= head.out_features for value in token_ids): + raise ValueError("token ID outside vocabulary") + indices = torch.tensor(token_ids, dtype=torch.long, device=head.weight.device) + else: + raise TypeError("token_ids must be a list, tuple or int64 tensor") + if not indices.numel(): + raise ValueError("at least one candidate token is required") + if bool(((indices < 0) | (indices >= head.out_features)).any()): + raise ValueError("token ID outside vocabulary") + weight = head.weight.index_select(0, indices) + bias = None if head.bias is None else head.bias.index_select(0, indices) + return F.linear(hidden_states, weight, bias) diff --git a/4B-5949/openjet_runtime/contracts.py b/4B-5949/openjet_runtime/contracts.py new file mode 100644 index 0000000000000000000000000000000000000000..8ce59a0564743435043ecc36a0577a16bfd42edc --- /dev/null +++ b/4B-5949/openjet_runtime/contracts.py @@ -0,0 +1,102 @@ +"""Small shared contract. Prompts use a strict whitelist of input fields.""" + +import json +import math + +PROMPT_VERSION = "jev.dynamic.prompt.v2" +LABELS = tuple("ABCDEFGHIJKLMNOP") +BINARY_CRITERIA = [ + {"id": "yes", "description": "The stated proposition is true."}, + {"id": "no", "description": "The stated proposition is false."}, +] + + +def validate_request(record): + for key in ("id", "group_id", "state", "instructions"): + if not isinstance(record.get(key), str) or not record[key].strip(): + raise ValueError(f"{key} must be a nonempty string") + if record.get("primitive") not in ("choice", "noul", "score_level"): + raise ValueError("unsupported primitive") + criteria = record.get("criteria") + if not isinstance(criteria, list) or not 2 <= len(criteria) <= len(LABELS): + raise ValueError("criteria must contain 2..16 candidates") + ids = [] + for candidate in criteria: + if not isinstance(candidate, dict): + raise TypeError("candidate must be an object") + for key in ("id", "description"): + if not isinstance(candidate.get(key), str) or not candidate[key].strip(): + raise ValueError(f"candidate {key} must be nonempty") + ids.append(candidate["id"]) + if len(set(ids)) != len(ids): + raise ValueError("duplicate candidate ids") + if record["primitive"] != "choice" and criteria != BINARY_CRITERIA: + raise ValueError("noul and score_level require canonical yes/no criteria") + + +def validate_record(record): + validate_request(record) + if record.get("gold") not in [c["id"] for c in record["criteria"]]: + raise ValueError("gold must be a candidate id") + if not isinstance(record.get("provenance"), dict): + raise TypeError("provenance must be an object") + + +def label_mapping(record): + validate_request(record) + return dict(zip(LABELS, (c["id"] for c in record["criteria"]))) + + +def render_prompt_parts(record): + """Text prefix/suffix; callers MUST check tokenizer boundary equivalence.""" + validate_request(record) + prefix = "Shared state:\n" + record["state"] + "\n\n" + task = { + "primitive": record["primitive"], + "instructions": record["instructions"], + "criteria": [ + {"label": label, "description": candidate["description"]} + for label, candidate in zip(LABELS, record["criteria"]) + ], + } + suffix = json.dumps(task, ensure_ascii=False, sort_keys=True) + suffix += ( + "\nReturn only the selected letter: " + + ", ".join(LABELS[: len(record["criteria"])]) + + ".\nAnswer:" + ) + return prefix, suffix + + +def render_prompt(record): + return "".join(render_prompt_parts(record)) + + +def to_messages(record): + validate_record(record) + inverse = {candidate: label for label, candidate in label_mapping(record).items()} + return { + "messages": [ + {"role": "user", "content": render_prompt(record)}, + {"role": "assistant", "content": inverse[record["gold"]]}, + ] + } + + +def format_response(record, probabilities): + """Map ordered candidate probabilities; score_level is NOT aggregate Score.""" + mapping = label_mapping(record) + values = list(probabilities) + if len(values) != len(mapping) or any( + not math.isfinite(p) or p < 0 or p > 1 for p in values + ): + raise ValueError("invalid probabilities") + if not math.isclose(sum(values), 1, abs_tol=1e-5): + raise ValueError("probabilities must sum to one") + distribution = dict(zip(mapping.values(), values)) + result = {"type": record["primitive"], "probabilities": distribution} + if record["primitive"] == "choice": + result["choice"] = max(distribution, key=distribution.get) + else: + result["yes_probability"] = distribution["yes"] + return result diff --git a/4B-5949/openjet_runtime/early_exit.py b/4B-5949/openjet_runtime/early_exit.py new file mode 100644 index 0000000000000000000000000000000000000000..ebe0e9f2d75fb73c0175ffe3494b7c768380e39d --- /dev/null +++ b/4B-5949/openjet_runtime/early_exit.py @@ -0,0 +1,209 @@ +"""Actual Q1 no-cache layer-prefix execution for native Qwen3.5 decisions. + +Mirrors the mask/position preparation of Transformers Qwen3_5TextModel (5.16.1). +This is a version-audited reference, not a generic model or cache implementation. +""" + +from dataclasses import dataclass, replace + +import torch +from torch import Tensor, nn +from transformers.masking_utils import ( + create_causal_mask, + create_recurrent_attention_mask, +) + +from .candidate_projection import ( + UnsupportedCandidateHead, + candidate_logits, +) + + +@dataclass(frozen=True) +class DepthContinuation: + hidden: Tensor # complete sequence residual, BEFORE final norm + position_ids: Tensor + position_embeddings: tuple[Tensor, Tensor] + masks: dict[str, Tensor | None] + depth: int + owner: object + model_signature: tuple + + +@dataclass +class DepthDecision: + depth: int + logits: Tensor # [1,C] + projection_mode: str + + +class QwenEarlyExit(nn.Module): + """Begin once, stop at a real depth, optionally continue without replay. + + Q1 means one unpadded complete input sequence. No cache or token generation. + Continuations are ephemeral: do not mutate parameters/train-mode between + begin/advance/readout, or persist them across optimizer steps. + """ + + def __init__(self, model: nn.Module): + super().__init__() + self.model = model + if self.base.config.model_type not in {"qwen3_5", "qwen3_5_text"}: + raise ValueError("only Qwen3.5 text/conditional models are supported") + if len(self.backbone.layers) != self.backbone.config.num_hidden_layers: + raise ValueError("layer count/config mismatch") + if not set(self.backbone.config.layer_types) <= { + "linear_attention", + "full_attention", + }: + raise ValueError("unsupported hybrid layer type") + self._owner = object() + + @property + def base(self): + return ( + self.model.get_base_model() + if hasattr(self.model, "get_base_model") + else self.model + ) + + @property + def backbone(self): + return ( + self.base.model.language_model + if self.base.config.model_type == "qwen3_5" + else self.base.model + ) + + @property + def full_depth(self): + return len(self.backbone.layers) + + def _signature(self): + # Reference guard: optimizer updates and mode/device changes invalidate + # all outstanding continuations. No .data mutation is supported. + return ( + tuple((id(module), module.training) for module in self.model.modules()), + tuple( + (id(parameter), parameter._version, parameter.device, parameter.dtype) + for parameter in self.model.parameters() + ), + ) + + def begin(self, input_ids: Tensor) -> DepthContinuation: + if ( + input_ids.dtype != torch.long + or input_ids.ndim != 2 + or input_ids.shape[0] != 1 + or input_ids.shape[1] < 1 + ): + raise ValueError("Q1 requires nonempty int64 input_ids [1,S], no padding") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.base.config, name, None) + if token is not None and (input_ids == token).any(): + raise ValueError("multimodal placeholders are unsupported") + embedding = self.base.get_input_embeddings() + if ((input_ids < 0) | (input_ids >= embedding.weight.shape[0])).any(): + raise ValueError("input token outside vocabulary") + hidden = embedding(input_ids) + positions = ( + torch.arange(input_ids.shape[1], device=hidden.device) + .view(1, 1, -1) + .expand(4, 1, -1) + ) + text_positions = positions[0] + kwargs = { + "config": self.backbone.config, + "inputs_embeds": hidden, + "attention_mask": None, + "past_key_values": None, + "position_ids": text_positions, + } + masks = { + "full_attention": create_causal_mask(**kwargs), + "linear_attention": create_recurrent_attention_mask(**kwargs), + } + rotary = self.backbone.rotary_emb(hidden, positions[1:]) + return DepthContinuation( + hidden, text_positions, rotary, masks, 0, self._owner, self._signature() + ) + + def _check_state(self, state): + if state.owner is not self._owner: + raise ValueError("continuation belongs to a different executor") + if state.model_signature != self._signature(): + raise ValueError( + "stale continuation: model parameters or training mode changed" + ) + + def advance(self, state: DepthContinuation, target_depth: int) -> DepthContinuation: + self._check_state(state) + if ( + type(target_depth) is not int + or not state.depth < target_depth <= self.full_depth + ): + raise ValueError("target depth must advance within the model") + hidden = state.hidden + for index in range(state.depth, target_depth): + hidden = self.backbone.layers[index]( + hidden, + position_embeddings=state.position_embeddings, + attention_mask=state.masks[self.backbone.config.layer_types[index]], + position_ids=state.position_ids, + past_key_values=None, + use_cache=False, + ) + return replace(state, hidden=hidden, depth=target_depth) + + def readout( + self, state: DepthContinuation, candidate_token_ids: Tensor + ) -> DepthDecision: + self._check_state(state) + if state.depth < 1: + raise ValueError("readout requires at least one executed layer") + head = self.base.get_output_embeddings() + if ( + candidate_token_ids.dtype != torch.long + or candidate_token_ids.ndim != 1 + or candidate_token_ids.numel() < 2 + ): + raise ValueError("require at least two int64 candidate tokens [C]") + if candidate_token_ids.device != state.hidden.device: + raise ValueError("candidate IDs must share hidden device") + if candidate_token_ids.unique().numel() != candidate_token_ids.numel(): + raise ValueError("duplicate candidate token") + if ( + (candidate_token_ids < 0) | (candidate_token_ids >= head.weight.shape[0]) + ).any(): + raise ValueError("candidate token outside vocabulary") + # Never replace the residual used by continuation with normalized hidden. + hidden = self.backbone.norm(state.hidden[:, -1]) + try: + logits = candidate_logits(head, hidden, candidate_token_ids) + mode = "candidate_rows" + except UnsupportedCandidateHead: + # Preserve adapters/hooks/parametrizations by executing the real head. + logits = head(hidden).index_select(-1, candidate_token_ids) + mode = "full_head_fallback" + return DepthDecision(state.depth, logits, mode) + + def forward( + self, input_ids: Tensor, candidate_token_ids: Tensor, *, depths: tuple[int, ...] + ): + if ( + not depths + or any(type(d) is not int or not 1 <= d <= self.full_depth for d in depths) + or list(depths) != sorted(set(depths)) + ): + raise ValueError("depths must be strictly increasing valid layer counts") + state = self.begin(input_ids) + decisions = [] + for depth in depths: + state = self.advance(state, depth) + decisions.append(self.readout(state, candidate_token_ids)) + return tuple(decisions) diff --git a/4B-5949/openjet_runtime/runtime.py b/4B-5949/openjet_runtime/runtime.py new file mode 100644 index 0000000000000000000000000000000000000000..09839883299e5d0d2359df5612ecc0f264366bdf --- /dev/null +++ b/4B-5949/openjet_runtime/runtime.py @@ -0,0 +1,176 @@ +"""Portable, single-request merged Qwen3.5 decision reference runtime.""" + +import json +from pathlib import Path + +import torch +import transformers + +from .contracts import PROMPT_VERSION, format_response, label_mapping, render_prompt +from .early_exit import QwenEarlyExit + + +class OpenJet: + """Explicit low/high, not automatic routing. Text-only; never truncates.""" + + def __init__(self, model, tokenizer, depth_config, max_length=8192): + if transformers.__version__ != "5.16.1": + raise RuntimeError("Layer execution is audited for transformers==5.16.1") + self.model = model.eval() + self.tokenizer = tokenizer + self.wrapper = QwenEarlyExit(self.model) + self.device = self.model.get_input_embeddings().weight.device + self.max_length = max_length + if type(max_length) is not int or not 1 <= max_length <= 8192: + raise ValueError("max_length must be an integer in 1..8192") + if depth_config.get("prompt_version") != PROMPT_VERSION: + raise ValueError("checkpoint prompt version does not match runtime") + full = depth_config.get("full_depth") + low = depth_config.get("exit_depth") + if full != self.wrapper.full_depth: + raise ValueError("checkpoint depth and model depth disagree") + if type(low) is not int or not 0 < low < full: + raise ValueError("checkpoint does not declare a trained shallow exit") + if self.wrapper.backbone.config.layer_types[low - 1] != "full_attention": + raise ValueError("shallow exit must be a full-attention boundary") + self.depths = {"low": low, "high": full} + + @classmethod + def from_pretrained(cls, directory, device="cuda:0", dtype="bfloat16"): + """Load a local HF snapshot (download explicitly with a pinned revision).""" + from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer + + directory = Path(directory) + if (directory / "adapter_config.json").exists(): + raise ValueError("Expected a merged snapshot, not an adapter directory") + if dtype not in ("float32", "bfloat16"): + raise ValueError("supported dtypes: float32, bfloat16") + config = AutoConfig.from_pretrained(directory, local_files_only=True) + loader = AutoModelForCausalLM + if config.model_type == "qwen3_5": + from transformers import Qwen3_5ForConditionalGeneration + + loader = Qwen3_5ForConditionalGeneration + elif config.model_type != "qwen3_5_text": + raise ValueError("Expected Qwen3.5 text or conditional-generation model") + model = loader.from_pretrained( + directory, + local_files_only=True, + dtype=getattr(torch, dtype), + attn_implementation="sdpa", + ).to(device) + tokenizer = AutoTokenizer.from_pretrained(directory, local_files_only=True) + depth_config = json.loads((directory / "depth_config.json").read_text()) + return cls(model, tokenizer, depth_config) + + def _depth(self, effort): + if effort not in self.depths: + raise ValueError("effort must be low or high") + return self.depths[effort] + + def compile(self, request): + """Exact chat/no-thinking contract used in original native evaluation.""" + mapping = label_mapping(request) + prompt = self.tokenizer.apply_chat_template( + [{"role": "user", "content": render_prompt(request)}], + tokenize=False, + add_generation_prompt=True, + enable_thinking=False, + ) + ids = self.tokenizer.encode(prompt, add_special_tokens=False) + if not ids or len(ids) > self.max_length: + raise ValueError("input exceeds runtime limit; no truncation permitted") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.model.config, name, None) + if token is not None and token in ids: + raise ValueError("multimodal placeholders are unsupported") + tokens = [] + for label in mapping: + token = self.tokenizer.encode(label, add_special_tokens=False) + joint = self.tokenizer.encode(prompt + label, add_special_tokens=False) + if len(token) != 1 or joint != ids + token: + raise ValueError( + "candidate label is not single-token at answer boundary" + ) + if token[0] in self.tokenizer.all_special_ids: + raise ValueError("candidate label must not be special token") + tokens.append(token[0]) + if len(set(tokens)) != len(tokens): + raise ValueError("candidate token IDs must be unique") + return ids, tokens + + @torch.inference_mode() + def decide(self, request, effort="high"): + depth = self._depth(effort) + ids, candidates = self.compile(request) + input_ids = torch.tensor([ids], dtype=torch.long, device=self.device) + candidate_ids = torch.tensor(candidates, dtype=torch.long, device=self.device) + if effort == "high": + # Match the original reference high/full-vocabulary head path. + output = self.model( + input_ids=input_ids, + attention_mask=torch.ones_like(input_ids), + position_ids=torch.arange(len(ids), device=self.device).unsqueeze(0), + past_key_values=None, + use_cache=True, + return_dict=True, + logits_to_keep=1, + ) + logits = output.logits[0, -1].index_select(0, candidate_ids).float() + projection = "full_head" + else: + (decision,) = self.wrapper(input_ids, candidate_ids, depths=(depth,)) + logits = decision.logits[0].float() + projection = decision.projection_mode + response = format_response(request, logits.softmax(-1).tolist()) + response.update( + effort=effort, + executed_layers=depth, + prompt_tokens=len(ids), + logits=logits.tolist(), + projection=projection, + calibrated=False, + ) + return response + + @torch.inference_mode() + def generate_text(self, user_text, effort="high", max_new_tokens=128): + """TYPE greedy reference; replays prefix each token, not optimized serving.""" + depth = self._depth(effort) + if not isinstance(user_text, str) or not user_text.strip(): + raise ValueError("user_text must be nonempty") + if type(max_new_tokens) is not int or max_new_tokens < 1: + raise ValueError("max_new_tokens must be positive integer") + ids = self.tokenizer.apply_chat_template( + [{"role": "user", "content": user_text}], + tokenize=True, + add_generation_prompt=True, + enable_thinking=False, + return_dict=False, + ) + if not ids or len(ids) + max_new_tokens > self.max_length: + raise ValueError("prompt plus generation reservation exceeds limit") + eos = self.model.generation_config.eos_token_id + eos = [eos] if isinstance(eos, int) else list(eos or []) + generated = [] + for _ in range(max_new_tokens): + tensor = torch.tensor([ids + generated], device=self.device) + state = self.wrapper.advance(self.wrapper.begin(tensor), depth) + hidden = self.wrapper.backbone.norm(state.hidden[:, -1]) + token = self.model.get_output_embeddings()(hidden)[0].argmax().item() + generated.append(token) + if token in eos: + break + return { + "text": self.tokenizer.decode(generated, skip_special_tokens=True), + "token_ids": generated, + "effort": effort, + "executed_layers_per_token": depth, + "finish_reason": "eos" if generated[-1] in eos else "length", + "prompt_tokens": len(ids), + } diff --git a/4B-5949/processor_config.json b/4B-5949/processor_config.json new file mode 100644 index 0000000000000000000000000000000000000000..43c4343ec493f200e452f87a7c6649ecdfcce1ac --- /dev/null +++ b/4B-5949/processor_config.json @@ -0,0 +1,61 @@ +{ + "image_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_processor_type": "Qwen2VLImageProcessor", + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "merge_size": 2, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "size": { + "longest_edge": 16777216, + "shortest_edge": 65536 + }, + "temporal_patch_size": 2 + }, + "processor_class": "Qwen3VLProcessor", + "video_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "do_sample_frames": true, + "fps": 2, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "max_frames": 768, + "max_video_tokens": 768, + "merge_size": 2, + "min_frames": 4, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "return_metadata": false, + "size": { + "longest_edge": 25165824, + "shortest_edge": 4096 + }, + "temporal_patch_size": 2, + "video_processor_type": "Qwen3VLVideoProcessor" + } +} diff --git a/4B-5949/provenance/BASE-LICENSE.txt b/4B-5949/provenance/BASE-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/4B-5949/provenance/BASE-LICENSE.txt @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/4B-5949/provenance/SOURCE-PROJECT-LICENSE.txt b/4B-5949/provenance/SOURCE-PROJECT-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..261eeb9e9f8b2b4b0d119366dda99c6fd7d35c64 --- /dev/null +++ b/4B-5949/provenance/SOURCE-PROJECT-LICENSE.txt @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/4B-5949/provenance/code-license-provenance.json b/4B-5949/provenance/code-license-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..9d9d90c416a85a9210038a456ef1cb7d83253673 --- /dev/null +++ b/4B-5949/provenance/code-license-provenance.json @@ -0,0 +1,7 @@ +{ + "source_project": "xDAN-ms-swift-jev", + "source_paths_and_hashes": "source-provenance.json", + "scope": "Source provenance only; assembly does not choose or grant a new license for exported code, adapters or training data.", + "source_project_license_sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4", + "source_project_license_file": "SOURCE-PROJECT-LICENSE.txt" +} diff --git a/4B-5949/provenance/identity.json b/4B-5949/provenance/identity.json new file mode 100644 index 0000000000000000000000000000000000000000..2b8b7e49e1ae9352b80f5975d4011854f9c31e28 --- /dev/null +++ b/4B-5949/provenance/identity.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-4B", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "checkpoint_step": 5949, + "adapter_sha256": "71154f60ec72c55d2c6c6147b9cdda9cc9a52d46297f3074dded7cdc9f5bc344", + "adapter_config_sha256": "c6caee3818ca1f6c8539e47fac9b9fa818d28b61a4bbb05f6f8444c0dd639e45", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/4B-5949/release-manifest.json b/4B-5949/release-manifest.json new file mode 100644 index 0000000000000000000000000000000000000000..3d83ff0cd71de35c149199ced2ca89dcc2527779 --- /dev/null +++ b/4B-5949/release-manifest.json @@ -0,0 +1,193 @@ +{ + "format": "openjet.merged-release.v1", + "size": "4B", + "checkpoint_step": 5949, + "base_id": "Qwen/Qwen3.5-4B", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "numerical_comparison_passed": false, + "decision_parity_passed": true, + "decision_agreement": 160, + "artifact_reviewed_variant": false, + "assembly_stage": "final", + "weight_arithmetic_passed": true, + "runtime_validation_status": "passed", + "private_staging": true, + "fresh_hub_download_gpu_check": "not established by assembly; consult publication receipt", + "files": [ + { + "path": "LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "README.md", + "bytes": 1924, + "sha256": "4071763e7fbd1a16910c7bd8d7b6751e09b999f98bcdbe1d2f118f6b10f0c40f" + }, + { + "path": "RUNTIME.md", + "bytes": 5556, + "sha256": "d9397e883c8b8709297f394c7c65eb0ee6b17c0994902b6f6d518d924fd2c854" + }, + { + "path": "chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "config.json", + "bytes": 2829, + "sha256": "4b53bdb886bb68d228841409c2db2f5d7eb670c28ae4376849f58d3bd014968d" + }, + { + "path": "decision-release.json", + "bytes": 1486, + "sha256": "51f8cd280dae539bc8cf1055b6a6ae64b8dfb13064ec3d155cf40ebe9079c690" + }, + { + "path": "depth_config.json", + "bytes": 320, + "sha256": "50a4f97bbb3589222d81285ff94b33efd0f048d159166f6b93f2700e633be3e0" + }, + { + "path": "evaluation/runtime-smoke.json", + "bytes": 4858, + "sha256": "60a80902675971ff6a054377852b936eec90f49f466eb7b8b077eee7b979f230" + }, + { + "path": "evaluation/weight-arithmetic.json", + "bytes": 25495, + "sha256": "567c5c0f85e9b1db314c58c896542eaf0753d0723fa94e61f134660e813ea18d" + }, + { + "path": "examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "merge-provenance.json", + "bytes": 682, + "sha256": "3dcce1f04722cfd98a30c060b96eb1fc6b8b1d321904486acf1c1b2693c66d3a" + }, + { + "path": "merged-evaluation.json", + "bytes": 3399, + "sha256": "b653bb0672e530a01ac58ab1b4597f7a9c81ddc157db2b0fdc8cf64ae73d85ed" + }, + { + "path": "model-00001-of-00003.safetensors", + "bytes": 3991298872, + "sha256": "c04bef62040612a3376c144014c194687cdc19b18c3a807ee1136b4a713cbb0a" + }, + { + "path": "model-00002-of-00003.safetensors", + "bytes": 3979833152, + "sha256": "7891d5b76b686893680e9cc074c2e17a788ff0cb03f64cc2ae4150bd804299a2" + }, + { + "path": "model-00003-of-00003.safetensors", + "bytes": 1107487880, + "sha256": "a88eecd668f83cad773bd67c9c7e6e466c1746d489c55d906b48398a6679db65" + }, + { + "path": "model.safetensors.index.json", + "bytes": 66236, + "sha256": "cd67b86e2cd9d329167224ac0026be69b39e078ad3056fc59339ba0a19fe4845" + }, + { + "path": "openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "provenance/identity.json", + "bytes": 682, + "sha256": "3dcce1f04722cfd98a30c060b96eb1fc6b8b1d321904486acf1c1b2693c66d3a" + }, + { + "path": "requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "runtime-validation.json", + "bytes": 334, + "sha256": "96d38fd523772afe5cc96e645fa2bc14d20e58a1920da3822e6c42b6cfeaf3e4" + }, + { + "path": "source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "training.md", + "bytes": 2692, + "sha256": "30544dc7ca3119ff10e88507651a3275b62f61b7e888f3332eb8fab632caae6b" + } + ], + "family_documentation_update": { + "source_manifest_sha256": "9dd56a1d3ee0526a6a9f2136393188edfe3a61eda79febc00093d65c6ed1b787", + "scope": "README only; source weights, runtime and evaluation evidence unchanged" + }, + "organization_documentation_update": { + "source_repo": "gump2049/APUS-OpenJev-v1", + "source_revision": "e7e3cc0b9c82b91380ec6595120b7ce8abd32fd7", + "previous_manifest_sha256": "e7f96d2b30cc3296aa975efd7b3bf8721b97990937c523fe4d78d8c605199fce", + "scope": "README download links only; all model and runtime bytes unchanged" + } +} diff --git a/4B-5949/requirements.txt b/4B-5949/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..f015223d23acca43e86d5d1d8579ab7af82c2673 --- /dev/null +++ b/4B-5949/requirements.txt @@ -0,0 +1,5 @@ +# Install torch with the CUDA build matching the deployment host first. +torch==2.8.0 +transformers==5.16.1 +safetensors>=0.6 +huggingface_hub diff --git a/4B-5949/runtime-validation.json b/4B-5949/runtime-validation.json new file mode 100644 index 0000000000000000000000000000000000000000..70755cc9df695260f2e42d9d508d91a1bee1feb5 --- /dev/null +++ b/4B-5949/runtime-validation.json @@ -0,0 +1,10 @@ +{ + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "60a80902675971ff6a054377852b936eec90f49f466eb7b8b077eee7b979f230" +} diff --git a/4B-5949/source-provenance.json b/4B-5949/source-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..e0432c1e4fb95b284334f882ed41af550fce5989 --- /dev/null +++ b/4B-5949/source-provenance.json @@ -0,0 +1,17 @@ +{ + "contracts.py": { + "source": "jev/dynamic/contracts.py", + "source_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "export_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + "candidate_projection.py": { + "source": "jev/dynamic/engine/candidate_projection.py", + "source_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6", + "export_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "early_exit.py": { + "source": "jev/dynamic/native/early_exit.py", + "source_sha256": "e6ce9e983df0fdaaa9fb9bdf4a6e1f37e0610708007648b2f6b0b9d36e35d2e3", + "export_sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + } +} diff --git a/4B-5949/tokenizer.json b/4B-5949/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5520bfd2dd834ce386c1312c410fa71af56db5ad --- /dev/null +++ b/4B-5949/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523 +size 19989325 diff --git a/4B-5949/tokenizer_config.json b/4B-5949/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d134cd298be1e3be25db393d93a1cefe80e3214 --- /dev/null +++ b/4B-5949/tokenizer_config.json @@ -0,0 +1,33 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": true, + "local_files_only": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "processor_class": "Qwen3VLProcessor", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/4B-5949/training.md b/4B-5949/training.md new file mode 100644 index 0000000000000000000000000000000000000000..4e7819b639acc2cc0193ac06995cb82c1bff4dc1 --- /dev/null +++ b/4B-5949/training.md @@ -0,0 +1,24 @@ +# This release: 4B, step 5949 + +This is the completed 5949-step SFT endpoint. + +# Training provenance + +Both families use the registered 5949-record SFT curriculum (3898 parent groups), with 5949 maximum steps and one epoch. Each checkpoint's recorded training step, epoch, and original trainer-state hash are preserved in the separate [LoRA archive manifest](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/manifest.json) at fixed revision `1e5557f923746031f8187b7daf299b9bee41cb3c` (repository access required). This merged package's `release-manifest.json` inventories inference artifacts and does not contain that per-checkpoint trainer-state record; intermediate checkpoints did not finish the whole schedule. Randomized loader order means the step number alone is not a verified count of unique examples seen at an intermediate checkpoint. + +| Source | Scheduled records | Parent groups | +|---|---:|---:| +| Mind2Web browser Choice | 1798 | 671 | +| Mind2Web browser TYPE | 158 | 125 | +| HelpSteer3 principle | 1300 | 1300 | +| SGD | 513 | 10 | +| GoEmotions independent-attribute Score | 500 | 297 | +| BoolQ | 600 | 600 | +| MNLI | 1000 | 1000 | +| Local counterfactual | 80 | 20 | + +Parent groups can overlap across browser Choice/TYPE. The 4B run initialized from a 427-record pilot adapter (weight SHA256 `0047e5f1f0c98f17da93def94032041997592e1a609ddcab58de41f4d30e3a38`), while the inspected 9B config records no origin adapter. Do not add pilot records to the registered schedule as if all were independent. + +Decision objective: `0.5 CE(low) + 0.5 CE(high) + 0.1 KL(P_high.detach || P_low)`. TYPE examples use full-depth text cross-entropy. LoRA r=8, alpha=16, dropout=0; learning rate 1e-4; batch size 1; seed 20260920; training max length 6144. The recorded compiled schedule maximum is 5845 tokens. Training settings and data/schedule hashes are retained in the separate [LoRA checkpoint depth configuration](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/4b/checkpoint-5949/depth_config.json) at fixed archive revision `1e5557f923746031f8187b7daf299b9bee41cb3c`. The merged package's `depth_config.json` contains only portable inference and identity metadata; it is not the full training configuration. + +This is SFT with a within-model distillation term; it does not prove RLCD, online reinforcement learning or teacher-model OPD occurred. The declared public sources contain multiple licensing regimes (including CC-BY, CC-BY-SA and mixed-source material); separate provenance and redistribution review remains necessary. Raw datasets are not part of this upload. diff --git a/9B-3000/LICENSE b/9B-3000/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/9B-3000/LICENSE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/9B-3000/README.md b/9B-3000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0723affa61ecf49b43ef2d4631f2d3af9388b735 --- /dev/null +++ b/9B-3000/README.md @@ -0,0 +1,33 @@ +# APUS-OpenJev-v1 · 9B + +A decision model for browser agents and business workflows. This directory contains standalone BF16 weights and a runtime with selectable `effort="low"` and `effort="high"` compute budgets. + +[Model family](../README.md) · [Architecture](../ARCHITECTURE.md) · [Runtime guide](RUNTIME.md) + +## Quick start + +Use a CUDA-capable PyTorch environment. + +```bash +python -m pip install huggingface_hub +hf auth login +hf download apus-ailab/APUS-OpenJev-v1 \ + --include "9B-3000/*" --local-dir ./APUS-OpenJev-v1 +cd ./APUS-OpenJev-v1/9B-3000 +python -m pip install -r requirements.txt +python examples.py . --device cuda:0 --effort high +``` + +The included runtime provides compute-budget selection. Use `high` for text generation. + +## Evaluation + +With the full compute budget, this merged model scores **68/80 (85.00%)** on the [Frozen80 development panel](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921): browser action selection, principle-based judgment, evidence-based questions, natural language inference, and attribute decisions. + +This reused development panel is an engineering reference, not an independent benchmark. BF16 merging changes some candidate probabilities; decision thresholds require revalidation. See [evaluation results](merged-evaluation.json) and [runtime checks](evaluation/runtime-smoke.json) for details. + +## Provenance + +We thank the Qwen team for the [Qwen3.5-9B](https://huggingface.co/Qwen/Qwen3.5-9B) base model. Training details and source records are in [training.md](training.md); artifact hashes are in [release-manifest.json](release-manifest.json). Consult [LICENSE](LICENSE) and the [base-model](provenance/BASE-LICENSE.txt) and [source-project](provenance/SOURCE-PROJECT-LICENSE.txt) notices. + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab) diff --git a/9B-3000/RUNTIME.md b/9B-3000/RUNTIME.md new file mode 100644 index 0000000000000000000000000000000000000000..e6e76c9d13b52c2f3a872136072081aacdd05404 --- /dev/null +++ b/9B-3000/RUNTIME.md @@ -0,0 +1,55 @@ +# xDAN-openJet merged reference runtime + +本目录是可随 HF merged 仓库发布的完整 Python 源码闭包,不需要安装 ms-swift、PEFT 或原项目。发布选择为 **4B checkpoint-5949**、**9B checkpoint-3000** 与 **9B checkpoint-5949**;各目录必须保留自己的 `depth_config.json`、完整 Qwen config、tokenizer、chat template、generation config 和 merged safetensors。不得将不同 checkpoint 的权重或 shallow/full 结果拼在一起。 + +## 运行 + +使用 CUDA 对应 PyTorch 2.8.0 构建,安装 `requirements.txt`。运行依赖 Transformer 私有模型层接口,所以严格要求 `transformers==5.16.1`;版本升级需重做层级及数值验证。这是原生 PyTorch 单卡单请求参考实现,不是 vLLM 服务。 + +先把发布仓库的**固定 commit**完整下载到本机目录。仓库根目录有本目录中的 `openjet_runtime/` 与 `examples.py` 时: + +```bash +python -m pip install -r requirements.txt +python examples.py ./model-snapshot --device cuda:0 --effort both +python examples.py ./model-snapshot --device cuda:0 --effort high --text +``` + +若代码与权重同在下载快照根目录,进入快照后把 `./model-snapshot` 改为 `.`。 + +```python +from openjet_runtime import OpenJet +from examples import decision_examples + +model = OpenJet.from_pretrained("./model-snapshot") +two_candidates, sixteen_candidates = decision_examples() +print(model.decide(two_candidates, effort="low")) +print(model.decide(sixteen_candidates, effort="high")) +print(model.generate_text("Return only the text: red shoes", effort="high", max_new_tokens=32)) +``` + +`examples.py` 中两候选工作流、16 候选浏览器和 TYPE 是接口演示,不是声称模型已通过的 benchmark。需要针对业务设计 prompt 和验证答案。 + +## 接口和执行语义 + +- `decide(request, effort)` 输入字段与原 `jev.dynamic.prompt.v2` 相同:`id/group_id/state/instructions/primitive/criteria`;每个候选有非空 `id` 和 `description`,2–16 个,ID 不重复。标签为 A–P,编译器验证每个标签在真实回答边界恰为一个 token。没有 gold 输入需求。 +- `primitive="choice"` 返回 `choice` 与按输入顺序映射的 `probabilities`;`noul/score_level` 使用 `contracts.py` 中固定 Yes/No 候选,返回 `yes_probability`。`score_level` 是单个命题的判断,不能当作完整序数 Score API。 +- `effort="low"` 执行 `depth_config.exit_depth`(这两项发布预期16),共享原 LM final norm 与候选行投影;`high` 执行 `full_depth`(预期32),保持原评测中的标准模型前向+完整 LM head 路径。读取 config,不凭参数规模推测层数。 +- 所有输入采用 tokenizer 自带 chat template、`enable_thinking=False`,超过8192 token直接报错。不会默默截断 state、instructions 或候选。 +- `probabilities` 是在当前候选集上的相对 softmax,**未经概率校准**;合并不自动带来可信置信度或校准保证。 +- `generate_text` 为 TYPE 文本保留的贪心参考路径;每个 token 重新计算前缀,方便与原评测逐 token 核对,但不适合宣传 tokens/s。达到上限明确返回 `finish_reason="length"`。 +- `both` 示例分别调用两个 effort;没有自动路由、不承诺共享两次调用的前缀缓存。本次便携发布不包含 KV 广播引擎、vLLM 插件、TypeSafe HTTP server 或多模态输入能力。 + +## 合并验收(GPU,不能用静态测试代替) + +1. 固定同一 base revision、adapter SHA、tokenizer、chat template、dtype、attention backend 和 Transformers 版本,记录 merge 前后权重身份。保留 adapter 原文件。 +2. 同进程/新进程分别加载 base+adapter 和 merged;比较固定2候选、16候选、长输入、80题面板两 effort 的 token IDs、候选排序、logits、probabilities 和最终ID。保存逐题差异与最大绝对差,不能只比 aggregate accuracy。 +3. BF16 merge 会舍入:不能预先声明 bitwise一致或把漂移简单解释为无害;应报告数值误差和所有预测翻转。若超过事先制定的容忍值,停止发布数值等价结论,考虑 FP32 merge/存储再独立评测。 +4. fresh reload 验证 `depth_config` 与实际层数、模块边界一致。用层 forward hooks 检查 low只执行浅层、high执行全层;hooks会触发候选头保守fallback,不拿该测量做性能报告。 +5. TYPE短样本比较 token序列和EOS;分别测试空输入、非法候选/重复ID、超长输入明确失败。 +6. 在干净环境固定 HF revision 下载,运行本目录例子和同一小面板。记录显存、依赖、GPU型号以及权重checksum。成功加载只是第一关,不能当作质量或吞吐验收。 + +原始数值等价门结果:`False`,保留原结果;决策完全一致门:`True`,一致 `160/160`。独立运行时 GPU 验收状态:`passed`。若数值门失败,此 BF16 包作为独立重评版本,禁止直接迁移概率/拒答/路由阈值;见 `merged-evaluation.json`。 + +## 源码来历 + +`contracts.py`、`candidate_projection.py` 与 `early_exit.py` 从已有本地实现原样提取(最后一项仅调整相对 import);`source-provenance.json` 记录源路径与两端 SHA256。`runtime.py` 是最小加载及接口层;high 与 TYPE 分别对应原 `HFDecisionEngine._forward_batch` 和 `package_eval.prefix_next_token` 的执行语义。采用现有模型类:`qwen3_5` → `Qwen3_5ForConditionalGeneration`;`qwen3_5_text` → `AutoModelForCausalLM`。没有新增学习参数。 diff --git a/9B-3000/chat_template.jinja b/9B-3000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..a585dec894e63da457d9440ec6aa7caa16d20860 --- /dev/null +++ b/9B-3000/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/9B-3000/config.json b/9B-3000/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d9f290f1cb06a1e0e50ea151676447f0a59bd4a3 --- /dev/null +++ b/9B-3000/config.json @@ -0,0 +1,109 @@ +{ + "architectures": [ + "Qwen3_5ForConditionalGeneration" + ], + "dtype": "bfloat16", + "image_token_id": 248056, + "model_type": "qwen3_5", + "text_config": { + "attention_bias": false, + "attention_dropout": 0.0, + "attn_output_gate": true, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 248044, + "full_attention_interval": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 12288, + "layer_types": [ + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention" + ], + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 16, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "mamba_ssm_dtype": "float32", + "max_position_embeddings": 262144, + "mlp_only_layers": [], + "model_type": "qwen3_5_text", + "mtp_num_hidden_layers": 1, + "mtp_use_dedicated_embeddings": false, + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 4, + "pad_token_id": null, + "partial_rotary_factor": 0.25, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "mrope_interleaved": true, + "mrope_section": [ + 11, + 11, + 10 + ], + "partial_rotary_factor": 0.25, + "rope_theta": 10000000, + "rope_type": "default" + }, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 248320 + }, + "tie_word_embeddings": false, + "transformers_version": "5.16.1", + "video_token_id": 248057, + "vision_config": { + "deepstack_visual_indexes": [], + "depth": 27, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "in_channels": 3, + "initializer_range": 0.02, + "intermediate_size": 4304, + "model_type": "qwen3_5_vision", + "num_heads": 16, + "num_position_embeddings": 2304, + "out_hidden_size": 4096, + "patch_size": 16, + "spatial_merge_size": 2, + "temporal_patch_size": 2 + }, + "vision_end_token_id": 248054, + "vision_start_token_id": 248053 +} diff --git a/9B-3000/decision-release.json b/9B-3000/decision-release.json new file mode 100644 index 0000000000000000000000000000000000000000..d088890f25ea31b5a77e223f9e5ead964b46b12e --- /dev/null +++ b/9B-3000/decision-release.json @@ -0,0 +1,35 @@ +{ + "passed": true, + "scope": "Independent BF16 merged variant, frozen80 decision parity; NOT a probability-equivalent replacement", + "post_observation_scope_amendment": true, + "original_numerical_gate_passed": false, + "original_gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "original_comparison_sha256": "d2ea8a7af10f09fdfde166384ffb18fb5a19ebf1a1281c58cb72806a45878c41", + "decision_agreement": 160, + "total_decisions": 160, + "independent_questions": 80, + "probability_calibration_transfer_validated": false, + "automatic_routing_validated": false, + "required_followup": "Recalibrate all probability, rejection and routing thresholds; independently evaluate new tasks", + "depths": { + "16": { + "before_correct": 63, + "after_correct": 63, + "changed_decisions": [], + "max_probability_abs_difference": 0.08810508251190186, + "mean_probability_abs_difference": 0.0019464968157012663, + "max_logit_abs_difference": 0.265625, + "mean_logit_abs_difference": 0.02963104248046875 + }, + "32": { + "before_correct": 68, + "after_correct": 68, + "changed_decisions": [], + "max_probability_abs_difference": 0.062176525592803955, + "mean_probability_abs_difference": 0.001304240933347387, + "max_logit_abs_difference": 0.25, + "mean_logit_abs_difference": 0.03828125 + } + }, + "publication_requires": "Further portable-runtime160, weight arithmetic, full file hashes and fresh HF reload gates" +} diff --git a/9B-3000/depth_config.json b/9B-3000/depth_config.json new file mode 100644 index 0000000000000000000000000000000000000000..e2fa0dfd4fbcb885175a7e6c16c18c286136a852 --- /dev/null +++ b/9B-3000/depth_config.json @@ -0,0 +1,10 @@ +{ + "prompt_version": "jev.dynamic.prompt.v2", + "exit_depth": 16, + "full_depth": 32, + "model_series": "xDAN-openJet", + "checkpoint_step": 3000, + "source_depth_config_sha256": "b6cab8011c151a7a5a5fce574addd8a5eb3d20e14cb516a1fb80d69563b4b4c9", + "training_mode": "two_exit", + "automatic_routing_validated": false +} diff --git a/9B-3000/evaluation/runtime-smoke.json b/9B-3000/evaluation/runtime-smoke.json new file mode 100644 index 0000000000000000000000000000000000000000..09a0e68bc8756b3d520d6b207af20d35322c1f02 --- /dev/null +++ b/9B-3000/evaluation/runtime-smoke.json @@ -0,0 +1,187 @@ +{ + "status": "passed", + "runtime_sha256": { + "openjet_runtime/__init__.py": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944", + "openjet_runtime/runtime.py": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c", + "openjet_runtime/early_exit.py": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566", + "openjet_runtime/contracts.py": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "openjet_runtime/candidate_projection.py": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "synthetic_examples": [ + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.9984512329101562, + "refund": 0.0015487611526623368 + }, + "choice": "close", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 103, + "logits": [ + 7.8125, + 1.34375 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.999480664730072, + "refund": 0.0005193048273213208 + }, + "choice": "close", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 103, + "logits": [ + 20.75, + 13.1875 + ], + "projection": "full_head", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 0.0071670678444206715, + "click-2": 0.0058954693377017975, + "click-3": 0.07841967791318893, + "click-4": 0.02626730129122734, + "click-5": 0.012456356547772884, + "click-6": 0.05142887309193611, + "click-7": 0.032435785979032516, + "click-8": 0.02796139568090439, + "click-9": 0.031437840312719345, + "click-10": 0.031437840312719345, + "click-11": 0.08090898394584656, + "click-12": 0.11409995704889297, + "click-13": 0.10887493193149567, + "click-14": 0.17128431797027588, + "click-15": 0.05648357421159744, + "click-16": 0.16344064474105835 + }, + "choice": "click-14", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 383, + "logits": [ + -0.158203125, + -0.353515625, + 2.234375, + 1.140625, + 0.39453125, + 1.8125, + 1.3515625, + 1.203125, + 1.3203125, + 1.3203125, + 2.265625, + 2.609375, + 2.5625, + 3.015625, + 1.90625, + 2.96875 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 0.0001901919167721644, + "click-2": 0.00011535723751876503, + "click-3": 0.00012279713700991124, + "click-4": 0.0001901919167721644, + "click-5": 9.563451021676883e-05, + "click-6": 8.984030864667147e-05, + "click-7": 5.4490898037329316e-05, + "click-8": 0.0001391473924741149, + "click-9": 0.0005503385909833014, + "click-10": 0.0001391473924741149, + "click-11": 0.0024664464872330427, + "click-12": 0.9950355291366577, + "click-13": 0.00033379721571691334, + "click-14": 0.00017866877897176892, + "click-15": 0.00010836809815373272, + "click-16": 0.0001901919167721644 + }, + "choice": "click-12", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 383, + "logits": [ + 12.4375, + 11.9375, + 12.0, + 12.4375, + 11.75, + 11.6875, + 11.1875, + 12.125, + 13.5, + 12.125, + 15.0, + 21.0, + 13.0, + 12.375, + 11.875, + 12.4375 + ], + "projection": "full_head", + "calibrated": false + } + ], + "text_smoke": { + "low": { + "text": "nesssss\u5730\u4e2d\u56fd\u5bb6\u5730...\u2026sX\u5d07\u5cf0\u5cf0...\u2026", + "token_ids": [ + 248068, + 2022, + 82, + 753, + 95852, + 106947, + 95852, + 1076, + 1873, + 82, + 55, + 98390, + 97462, + 97462, + 1076, + 1873 + ], + "effort": "low", + "executed_layers_per_token": 16, + "finish_reason": "length", + "prompt_tokens": 28 + }, + "high": { + "text": "red shoes\n", + "token_ids": [ + 1114, + 14850, + 248046, + 198, + 248044 + ], + "effort": "high", + "executed_layers_per_token": 32, + "finish_reason": "eos", + "prompt_tokens": 28 + } + }, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation" +} diff --git a/9B-3000/evaluation/weight-arithmetic.json b/9B-3000/evaluation/weight-arithmetic.json new file mode 100644 index 0000000000000000000000000000000000000000..eb24ac6cc55833e46acd84282d14b01f92d2deb6 --- /dev/null +++ b/9B-3000/evaluation/weight-arithmetic.json @@ -0,0 +1,908 @@ +{ + "status": "passed", + "base_mtp_keys_not_loaded_by_transformers": [ + "mtp.fc.weight", + "mtp.layers.0.input_layernorm.weight", + "mtp.layers.0.mlp.down_proj.weight", + "mtp.layers.0.mlp.gate_proj.weight", + "mtp.layers.0.mlp.up_proj.weight", + "mtp.layers.0.post_attention_layernorm.weight", + "mtp.layers.0.self_attn.k_norm.weight", + "mtp.layers.0.self_attn.k_proj.weight", + "mtp.layers.0.self_attn.o_proj.weight", + "mtp.layers.0.self_attn.q_norm.weight", + "mtp.layers.0.self_attn.q_proj.weight", + "mtp.layers.0.self_attn.v_proj.weight", + "mtp.norm.weight", + "mtp.pre_fc_norm_embedding.weight", + "mtp.pre_fc_norm_hidden.weight" + ], + "size": "9B", + "targeted_weight_matrices": 176, + "adapter_tensors": 352, + "all_targeted_weights_exact": true, + "scope": "FP32 LoRA arithmetic followed by BF16 rounding; not forward/probability equivalence", + "adapter_sha256": "14dd3cbaa26ced2ba5237dfff3aad93af4350dd9a43b869c7ce62cc9dd38d03b", + "checks": [ + { + "weight": "model.language_model.layers.0.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + } + ] +} diff --git a/9B-3000/examples.py b/9B-3000/examples.py new file mode 100644 index 0000000000000000000000000000000000000000..c942c70fb5ee82412ae703b04eea3ea58d4226b8 --- /dev/null +++ b/9B-3000/examples.py @@ -0,0 +1,61 @@ +"""Run against a local merged HF snapshot; no project-local dependencies.""" + +import argparse +import json + +from openjet_runtime import OpenJet + + +def decision_examples(): + binary = { + "id": "example-binary", + "group_id": "example-binary", + "primitive": "choice", + "state": "Order 731 has been delivered. The customer's message says thank you.", + "instructions": "Select the appropriate next workflow action.", + "criteria": [ + {"id": "close", "description": "Close the resolved support ticket."}, + {"id": "refund", "description": "Refund an undelivered order."}, + ], + } + browser = { + "id": "example-browser", + "group_id": "example-browser", + "primitive": "choice", + "state": "A settings page has 16 visible buttons labeled Page 1 through Page 16.", + "instructions": "Navigate to Page 12 by choosing its matching button.", + "criteria": [ + {"id": f"click-{i}", "description": f"Click the Page {i} button."} + for i in range(1, 17) + ], + } + return binary, browser + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("model", help="Local merged snapshot directory") + parser.add_argument("--device", default="cuda:0") + parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") + parser.add_argument("--effort", choices=["low", "high", "both"], default="both") + parser.add_argument( + "--text", action="store_true", help="Also run slow TYPE reference" + ) + args = parser.parse_args() + runtime = OpenJet.from_pretrained(args.model, args.device, args.dtype) + efforts = ("low", "high") if args.effort == "both" else (args.effort,) + for effort in efforts: + for request in decision_examples(): + result = runtime.decide(request, effort) + print(json.dumps({"example": request["id"], **result}, ensure_ascii=False)) + if args.text: + result = runtime.generate_text( + "Return only the literal text to type into a search box for 'red shoes'.", + effort=effort, + max_new_tokens=32, + ) + print(json.dumps({"example": "browser-type", **result}, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/9B-3000/generation_config.json b/9B-3000/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..f0b25ab7f068b3916a0b4e942812ee859e14c813 --- /dev/null +++ b/9B-3000/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "eos_token_id": 248044, + "transformers_version": "5.16.1", + "use_cache": true +} diff --git a/9B-3000/merge-provenance.json b/9B-3000/merge-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..2c4744609c27bea8e17c6be14c6d7ca35affe9dc --- /dev/null +++ b/9B-3000/merge-provenance.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "checkpoint_step": 3000, + "adapter_sha256": "14dd3cbaa26ced2ba5237dfff3aad93af4350dd9a43b869c7ce62cc9dd38d03b", + "adapter_config_sha256": "a3d03e9dfd4a3895d3a163ae679f931d957633deda178370ece7ebc86a1515ab", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/9B-3000/merged-evaluation.json b/9B-3000/merged-evaluation.json new file mode 100644 index 0000000000000000000000000000000000000000..b6a790c1e2bd2678f7182e3edb1a073f00456adc --- /dev/null +++ b/9B-3000/merged-evaluation.json @@ -0,0 +1,83 @@ +{ + "scope": "80 fixed requests, both explicit efforts; merged-model comparison, not independent generalization evidence", + "checkpoint_step": 3000, + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "comparison": { + "passed": false, + "gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "depths": { + "16": { + "before_correct": 63, + "after_correct": 63, + "changed_decisions": [], + "max_probability_abs_difference": 0.08810508251190186, + "mean_probability_abs_difference": 0.0019464968157012663, + "max_logit_abs_difference": 0.265625, + "mean_logit_abs_difference": 0.02963104248046875 + }, + "32": { + "before_correct": 68, + "after_correct": 68, + "changed_decisions": [], + "max_probability_abs_difference": 0.062176525592803955, + "mean_probability_abs_difference": 0.001304240933347387, + "max_logit_abs_difference": 0.25, + "mean_logit_abs_difference": 0.03828125 + } + }, + "before_sha256": "28412c70d8b456b35bb884a2fba4e85801b0135fc7957190927994a56d268a6f", + "after_sha256": "3f01a868d4fba91a8a1749df37a87f10f0874823eec5a978291c93231773c365", + "verified_at_unix": 1789984423.7235599 + }, + "runtime_validation": { + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "70d686828090965063966dddac6d8454cf981c6614a406c1647617986e534c5d" + }, + "decision_release": { + "passed": true, + "scope": "Independent BF16 merged variant, frozen80 decision parity; NOT a probability-equivalent replacement", + "post_observation_scope_amendment": true, + "original_numerical_gate_passed": false, + "original_gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "original_comparison_sha256": "d2ea8a7af10f09fdfde166384ffb18fb5a19ebf1a1281c58cb72806a45878c41", + "decision_agreement": 160, + "total_decisions": 160, + "independent_questions": 80, + "probability_calibration_transfer_validated": false, + "automatic_routing_validated": false, + "required_followup": "Recalibrate all probability, rejection and routing thresholds; independently evaluate new tasks", + "depths": { + "16": { + "before_correct": 63, + "after_correct": 63, + "changed_decisions": [], + "max_probability_abs_difference": 0.08810508251190186, + "mean_probability_abs_difference": 0.0019464968157012663, + "max_logit_abs_difference": 0.265625, + "mean_logit_abs_difference": 0.02963104248046875 + }, + "32": { + "before_correct": 68, + "after_correct": 68, + "changed_decisions": [], + "max_probability_abs_difference": 0.062176525592803955, + "mean_probability_abs_difference": 0.001304240933347387, + "max_logit_abs_difference": 0.25, + "mean_logit_abs_difference": 0.03828125 + } + }, + "publication_requires": "Further portable-runtime160, weight arithmetic, full file hashes and fresh HF reload gates" + }, + "reviewed_variant": {}, + "decision_parity_passed": true, + "decision_agreement": 160, + "weight_arithmetic_status": "passed", + "raw_inputs_included": false +} diff --git a/9B-3000/model-00001-of-00006.safetensors b/9B-3000/model-00001-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7a1670492c725c5be68f6ca47956025cbe4b1cf5 --- /dev/null +++ b/9B-3000/model-00001-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344 +size 2034237568 diff --git a/9B-3000/model-00002-of-00006.safetensors b/9B-3000/model-00002-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..470845bdb5c8e00e0663abd5345c956b29935576 --- /dev/null +++ b/9B-3000/model-00002-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fe0ee3d29088f7f84ac7ba8c9d7562b91e6a31841c5b585c71779e618519e99 +size 3999615808 diff --git a/9B-3000/model-00003-of-00006.safetensors b/9B-3000/model-00003-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7b49f1741cf2aafa7b8dd6744c65e79b798197da --- /dev/null +++ b/9B-3000/model-00003-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7bfce51b034a6de02c513b032c97007532676c5f915d020aa9a66f399f437117 +size 3997274128 diff --git a/9B-3000/model-00004-of-00006.safetensors b/9B-3000/model-00004-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..46d3fc5de66b9cb9e4aacbd30c2b696d5e6b4bb5 --- /dev/null +++ b/9B-3000/model-00004-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f36d2b3ffc4c44dab06277ebbd722f73faca80aefc9aa3ce1119575f4a4773ae +size 3997290904 diff --git a/9B-3000/model-00005-of-00006.safetensors b/9B-3000/model-00005-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..cc39f4d8ae199c56fc8547d07f57a563d4f52041 --- /dev/null +++ b/9B-3000/model-00005-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b06602342e98eb882703857f3c8e8058894c03d4f2d62ddaab1e6dd4913c4e9b +size 3991239264 diff --git a/9B-3000/model-00006-of-00006.safetensors b/9B-3000/model-00006-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..25da53d8687f76a008191a40e8d4c1829cdf221a --- /dev/null +++ b/9B-3000/model-00006-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013 +size 800062816 diff --git a/9B-3000/model.safetensors.index.json b/9B-3000/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..5e44268776a44b9751d1aae62d822136b8ecb6ce --- /dev/null +++ b/9B-3000/model.safetensors.index.json @@ -0,0 +1,768 @@ +{ + "metadata": { + "total_parameters": 9409813744, + "total_size": 18819627488 + }, + "weight_map": { + "lm_head.weight": "model-00001-of-00006.safetensors", + "model.language_model.embed_tokens.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.10.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.k_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.q_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.13.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.k_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.q_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.k_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.q_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.2.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.20.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.23.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.23.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.3.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.k_norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.q_norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.30.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.4.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.4.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.4.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.k_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.q_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.norm.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.10.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.2.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.20.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.norm.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.norm.weight": "model-00006-of-00006.safetensors", + "model.visual.patch_embed.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.patch_embed.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.pos_embed.weight": "model-00006-of-00006.safetensors" + } +} diff --git a/9B-3000/openjet_runtime/__init__.py b/9B-3000/openjet_runtime/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..b314e3fca191cba379bb38ac6156c1f57a93ba93 --- /dev/null +++ b/9B-3000/openjet_runtime/__init__.py @@ -0,0 +1,3 @@ +from .runtime import OpenJet + +__all__ = ["OpenJet"] diff --git a/9B-3000/openjet_runtime/candidate_projection.py b/9B-3000/openjet_runtime/candidate_projection.py new file mode 100644 index 0000000000000000000000000000000000000000..7241c2a9985c91f726c2a145a52a09a90a3e6449 --- /dev/null +++ b/9B-3000/openjet_runtime/candidate_projection.py @@ -0,0 +1,108 @@ +"""Project selected native dense LM-head rows before matmul, preserving autograd. + +The caller must pass the actual output head, never an unwrapped adapter base_layer. +Unsupported heads raise; there is deliberately no automatic full-head fallback. +""" + +import torch +from torch import nn +from torch.nn import functional as F +from torch.nn.modules import module as module_hooks + + +class UnsupportedCandidateHead(ValueError): + """The head's semantics cannot be reproduced by plain selected-row linear.""" + + +def _check_head(head): + if type(head) is not nn.Linear: + raise UnsupportedCandidateHead("only exact torch.nn.Linear is supported") + if ( + head.forward.__func__ is not nn.Linear.forward + if hasattr(head.forward, "__func__") + else True + ): + raise UnsupportedCandidateHead("overridden forward is unsupported") + if head._modules or head._buffers or set(head._parameters) != {"weight", "bias"}: + raise UnsupportedCandidateHead( + "head contains extra modules, buffers or parameters" + ) + for name in ( + "_forward_hooks", + "_forward_pre_hooks", + "_backward_hooks", + "_backward_pre_hooks", + ): + if getattr(head, name, None) or getattr(module_hooks, "_global" + name, None): + raise UnsupportedCandidateHead("module hooks would be bypassed") + for value in (head.weight, head.bias): + if value is None: + continue + if ( + type(value) is not nn.Parameter + or value.is_quantized + or value.layout != torch.strided + or not value.is_floating_point() + or value.device.type == "meta" + ): + raise UnsupportedCandidateHead( + "requires ordinary dense floating-point Parameters" + ) + if head.weight is None or head.weight.shape != ( + head.out_features, + head.in_features, + ): + raise UnsupportedCandidateHead("invalid dense weight shape") + if head.bias is not None and ( + head.bias.shape != (head.out_features,) + or head.bias.dtype != head.weight.dtype + or head.bias.device != head.weight.device + ): + raise UnsupportedCandidateHead( + "bias shape, dtype or device does not match weight" + ) + + +def candidate_logits(head, hidden_states, token_ids): + """Return [..., K] logits in supplied token order, including repeated IDs. + + Dtype/autocast follow F.linear; no explicit precision conversion or detach. + Shape-dependent floating GEMM rounding may differ from full-vocabulary GEMM. + This validates indices, not tokenization, semantic labels, or probability mass. + """ + _check_head(head) + if type(hidden_states) not in (torch.Tensor, nn.Parameter): + raise TypeError("hidden_states must be an ordinary Tensor") + if ( + hidden_states.ndim < 1 + or hidden_states.shape[-1] != head.in_features + or not hidden_states.is_floating_point() + or hidden_states.layout != torch.strided + or hidden_states.device != head.weight.device + ): + raise ValueError( + "hidden_states shape, floating layout or device does not match head" + ) + if type(token_ids) is torch.Tensor: + if ( + token_ids.ndim != 1 + or token_ids.dtype != torch.long + or token_ids.device.type == "meta" + ): + raise ValueError("token_ids must be a one-dimensional int64 tensor") + indices = token_ids.to(device=head.weight.device) + elif isinstance(token_ids, (list, tuple)): + if any(type(value) is not int for value in token_ids): + raise TypeError("token IDs must be integers, not booleans or floats") + if any(value < 0 or value >= head.out_features for value in token_ids): + raise ValueError("token ID outside vocabulary") + indices = torch.tensor(token_ids, dtype=torch.long, device=head.weight.device) + else: + raise TypeError("token_ids must be a list, tuple or int64 tensor") + if not indices.numel(): + raise ValueError("at least one candidate token is required") + if bool(((indices < 0) | (indices >= head.out_features)).any()): + raise ValueError("token ID outside vocabulary") + weight = head.weight.index_select(0, indices) + bias = None if head.bias is None else head.bias.index_select(0, indices) + return F.linear(hidden_states, weight, bias) diff --git a/9B-3000/openjet_runtime/contracts.py b/9B-3000/openjet_runtime/contracts.py new file mode 100644 index 0000000000000000000000000000000000000000..8ce59a0564743435043ecc36a0577a16bfd42edc --- /dev/null +++ b/9B-3000/openjet_runtime/contracts.py @@ -0,0 +1,102 @@ +"""Small shared contract. Prompts use a strict whitelist of input fields.""" + +import json +import math + +PROMPT_VERSION = "jev.dynamic.prompt.v2" +LABELS = tuple("ABCDEFGHIJKLMNOP") +BINARY_CRITERIA = [ + {"id": "yes", "description": "The stated proposition is true."}, + {"id": "no", "description": "The stated proposition is false."}, +] + + +def validate_request(record): + for key in ("id", "group_id", "state", "instructions"): + if not isinstance(record.get(key), str) or not record[key].strip(): + raise ValueError(f"{key} must be a nonempty string") + if record.get("primitive") not in ("choice", "noul", "score_level"): + raise ValueError("unsupported primitive") + criteria = record.get("criteria") + if not isinstance(criteria, list) or not 2 <= len(criteria) <= len(LABELS): + raise ValueError("criteria must contain 2..16 candidates") + ids = [] + for candidate in criteria: + if not isinstance(candidate, dict): + raise TypeError("candidate must be an object") + for key in ("id", "description"): + if not isinstance(candidate.get(key), str) or not candidate[key].strip(): + raise ValueError(f"candidate {key} must be nonempty") + ids.append(candidate["id"]) + if len(set(ids)) != len(ids): + raise ValueError("duplicate candidate ids") + if record["primitive"] != "choice" and criteria != BINARY_CRITERIA: + raise ValueError("noul and score_level require canonical yes/no criteria") + + +def validate_record(record): + validate_request(record) + if record.get("gold") not in [c["id"] for c in record["criteria"]]: + raise ValueError("gold must be a candidate id") + if not isinstance(record.get("provenance"), dict): + raise TypeError("provenance must be an object") + + +def label_mapping(record): + validate_request(record) + return dict(zip(LABELS, (c["id"] for c in record["criteria"]))) + + +def render_prompt_parts(record): + """Text prefix/suffix; callers MUST check tokenizer boundary equivalence.""" + validate_request(record) + prefix = "Shared state:\n" + record["state"] + "\n\n" + task = { + "primitive": record["primitive"], + "instructions": record["instructions"], + "criteria": [ + {"label": label, "description": candidate["description"]} + for label, candidate in zip(LABELS, record["criteria"]) + ], + } + suffix = json.dumps(task, ensure_ascii=False, sort_keys=True) + suffix += ( + "\nReturn only the selected letter: " + + ", ".join(LABELS[: len(record["criteria"])]) + + ".\nAnswer:" + ) + return prefix, suffix + + +def render_prompt(record): + return "".join(render_prompt_parts(record)) + + +def to_messages(record): + validate_record(record) + inverse = {candidate: label for label, candidate in label_mapping(record).items()} + return { + "messages": [ + {"role": "user", "content": render_prompt(record)}, + {"role": "assistant", "content": inverse[record["gold"]]}, + ] + } + + +def format_response(record, probabilities): + """Map ordered candidate probabilities; score_level is NOT aggregate Score.""" + mapping = label_mapping(record) + values = list(probabilities) + if len(values) != len(mapping) or any( + not math.isfinite(p) or p < 0 or p > 1 for p in values + ): + raise ValueError("invalid probabilities") + if not math.isclose(sum(values), 1, abs_tol=1e-5): + raise ValueError("probabilities must sum to one") + distribution = dict(zip(mapping.values(), values)) + result = {"type": record["primitive"], "probabilities": distribution} + if record["primitive"] == "choice": + result["choice"] = max(distribution, key=distribution.get) + else: + result["yes_probability"] = distribution["yes"] + return result diff --git a/9B-3000/openjet_runtime/early_exit.py b/9B-3000/openjet_runtime/early_exit.py new file mode 100644 index 0000000000000000000000000000000000000000..ebe0e9f2d75fb73c0175ffe3494b7c768380e39d --- /dev/null +++ b/9B-3000/openjet_runtime/early_exit.py @@ -0,0 +1,209 @@ +"""Actual Q1 no-cache layer-prefix execution for native Qwen3.5 decisions. + +Mirrors the mask/position preparation of Transformers Qwen3_5TextModel (5.16.1). +This is a version-audited reference, not a generic model or cache implementation. +""" + +from dataclasses import dataclass, replace + +import torch +from torch import Tensor, nn +from transformers.masking_utils import ( + create_causal_mask, + create_recurrent_attention_mask, +) + +from .candidate_projection import ( + UnsupportedCandidateHead, + candidate_logits, +) + + +@dataclass(frozen=True) +class DepthContinuation: + hidden: Tensor # complete sequence residual, BEFORE final norm + position_ids: Tensor + position_embeddings: tuple[Tensor, Tensor] + masks: dict[str, Tensor | None] + depth: int + owner: object + model_signature: tuple + + +@dataclass +class DepthDecision: + depth: int + logits: Tensor # [1,C] + projection_mode: str + + +class QwenEarlyExit(nn.Module): + """Begin once, stop at a real depth, optionally continue without replay. + + Q1 means one unpadded complete input sequence. No cache or token generation. + Continuations are ephemeral: do not mutate parameters/train-mode between + begin/advance/readout, or persist them across optimizer steps. + """ + + def __init__(self, model: nn.Module): + super().__init__() + self.model = model + if self.base.config.model_type not in {"qwen3_5", "qwen3_5_text"}: + raise ValueError("only Qwen3.5 text/conditional models are supported") + if len(self.backbone.layers) != self.backbone.config.num_hidden_layers: + raise ValueError("layer count/config mismatch") + if not set(self.backbone.config.layer_types) <= { + "linear_attention", + "full_attention", + }: + raise ValueError("unsupported hybrid layer type") + self._owner = object() + + @property + def base(self): + return ( + self.model.get_base_model() + if hasattr(self.model, "get_base_model") + else self.model + ) + + @property + def backbone(self): + return ( + self.base.model.language_model + if self.base.config.model_type == "qwen3_5" + else self.base.model + ) + + @property + def full_depth(self): + return len(self.backbone.layers) + + def _signature(self): + # Reference guard: optimizer updates and mode/device changes invalidate + # all outstanding continuations. No .data mutation is supported. + return ( + tuple((id(module), module.training) for module in self.model.modules()), + tuple( + (id(parameter), parameter._version, parameter.device, parameter.dtype) + for parameter in self.model.parameters() + ), + ) + + def begin(self, input_ids: Tensor) -> DepthContinuation: + if ( + input_ids.dtype != torch.long + or input_ids.ndim != 2 + or input_ids.shape[0] != 1 + or input_ids.shape[1] < 1 + ): + raise ValueError("Q1 requires nonempty int64 input_ids [1,S], no padding") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.base.config, name, None) + if token is not None and (input_ids == token).any(): + raise ValueError("multimodal placeholders are unsupported") + embedding = self.base.get_input_embeddings() + if ((input_ids < 0) | (input_ids >= embedding.weight.shape[0])).any(): + raise ValueError("input token outside vocabulary") + hidden = embedding(input_ids) + positions = ( + torch.arange(input_ids.shape[1], device=hidden.device) + .view(1, 1, -1) + .expand(4, 1, -1) + ) + text_positions = positions[0] + kwargs = { + "config": self.backbone.config, + "inputs_embeds": hidden, + "attention_mask": None, + "past_key_values": None, + "position_ids": text_positions, + } + masks = { + "full_attention": create_causal_mask(**kwargs), + "linear_attention": create_recurrent_attention_mask(**kwargs), + } + rotary = self.backbone.rotary_emb(hidden, positions[1:]) + return DepthContinuation( + hidden, text_positions, rotary, masks, 0, self._owner, self._signature() + ) + + def _check_state(self, state): + if state.owner is not self._owner: + raise ValueError("continuation belongs to a different executor") + if state.model_signature != self._signature(): + raise ValueError( + "stale continuation: model parameters or training mode changed" + ) + + def advance(self, state: DepthContinuation, target_depth: int) -> DepthContinuation: + self._check_state(state) + if ( + type(target_depth) is not int + or not state.depth < target_depth <= self.full_depth + ): + raise ValueError("target depth must advance within the model") + hidden = state.hidden + for index in range(state.depth, target_depth): + hidden = self.backbone.layers[index]( + hidden, + position_embeddings=state.position_embeddings, + attention_mask=state.masks[self.backbone.config.layer_types[index]], + position_ids=state.position_ids, + past_key_values=None, + use_cache=False, + ) + return replace(state, hidden=hidden, depth=target_depth) + + def readout( + self, state: DepthContinuation, candidate_token_ids: Tensor + ) -> DepthDecision: + self._check_state(state) + if state.depth < 1: + raise ValueError("readout requires at least one executed layer") + head = self.base.get_output_embeddings() + if ( + candidate_token_ids.dtype != torch.long + or candidate_token_ids.ndim != 1 + or candidate_token_ids.numel() < 2 + ): + raise ValueError("require at least two int64 candidate tokens [C]") + if candidate_token_ids.device != state.hidden.device: + raise ValueError("candidate IDs must share hidden device") + if candidate_token_ids.unique().numel() != candidate_token_ids.numel(): + raise ValueError("duplicate candidate token") + if ( + (candidate_token_ids < 0) | (candidate_token_ids >= head.weight.shape[0]) + ).any(): + raise ValueError("candidate token outside vocabulary") + # Never replace the residual used by continuation with normalized hidden. + hidden = self.backbone.norm(state.hidden[:, -1]) + try: + logits = candidate_logits(head, hidden, candidate_token_ids) + mode = "candidate_rows" + except UnsupportedCandidateHead: + # Preserve adapters/hooks/parametrizations by executing the real head. + logits = head(hidden).index_select(-1, candidate_token_ids) + mode = "full_head_fallback" + return DepthDecision(state.depth, logits, mode) + + def forward( + self, input_ids: Tensor, candidate_token_ids: Tensor, *, depths: tuple[int, ...] + ): + if ( + not depths + or any(type(d) is not int or not 1 <= d <= self.full_depth for d in depths) + or list(depths) != sorted(set(depths)) + ): + raise ValueError("depths must be strictly increasing valid layer counts") + state = self.begin(input_ids) + decisions = [] + for depth in depths: + state = self.advance(state, depth) + decisions.append(self.readout(state, candidate_token_ids)) + return tuple(decisions) diff --git a/9B-3000/openjet_runtime/runtime.py b/9B-3000/openjet_runtime/runtime.py new file mode 100644 index 0000000000000000000000000000000000000000..09839883299e5d0d2359df5612ecc0f264366bdf --- /dev/null +++ b/9B-3000/openjet_runtime/runtime.py @@ -0,0 +1,176 @@ +"""Portable, single-request merged Qwen3.5 decision reference runtime.""" + +import json +from pathlib import Path + +import torch +import transformers + +from .contracts import PROMPT_VERSION, format_response, label_mapping, render_prompt +from .early_exit import QwenEarlyExit + + +class OpenJet: + """Explicit low/high, not automatic routing. Text-only; never truncates.""" + + def __init__(self, model, tokenizer, depth_config, max_length=8192): + if transformers.__version__ != "5.16.1": + raise RuntimeError("Layer execution is audited for transformers==5.16.1") + self.model = model.eval() + self.tokenizer = tokenizer + self.wrapper = QwenEarlyExit(self.model) + self.device = self.model.get_input_embeddings().weight.device + self.max_length = max_length + if type(max_length) is not int or not 1 <= max_length <= 8192: + raise ValueError("max_length must be an integer in 1..8192") + if depth_config.get("prompt_version") != PROMPT_VERSION: + raise ValueError("checkpoint prompt version does not match runtime") + full = depth_config.get("full_depth") + low = depth_config.get("exit_depth") + if full != self.wrapper.full_depth: + raise ValueError("checkpoint depth and model depth disagree") + if type(low) is not int or not 0 < low < full: + raise ValueError("checkpoint does not declare a trained shallow exit") + if self.wrapper.backbone.config.layer_types[low - 1] != "full_attention": + raise ValueError("shallow exit must be a full-attention boundary") + self.depths = {"low": low, "high": full} + + @classmethod + def from_pretrained(cls, directory, device="cuda:0", dtype="bfloat16"): + """Load a local HF snapshot (download explicitly with a pinned revision).""" + from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer + + directory = Path(directory) + if (directory / "adapter_config.json").exists(): + raise ValueError("Expected a merged snapshot, not an adapter directory") + if dtype not in ("float32", "bfloat16"): + raise ValueError("supported dtypes: float32, bfloat16") + config = AutoConfig.from_pretrained(directory, local_files_only=True) + loader = AutoModelForCausalLM + if config.model_type == "qwen3_5": + from transformers import Qwen3_5ForConditionalGeneration + + loader = Qwen3_5ForConditionalGeneration + elif config.model_type != "qwen3_5_text": + raise ValueError("Expected Qwen3.5 text or conditional-generation model") + model = loader.from_pretrained( + directory, + local_files_only=True, + dtype=getattr(torch, dtype), + attn_implementation="sdpa", + ).to(device) + tokenizer = AutoTokenizer.from_pretrained(directory, local_files_only=True) + depth_config = json.loads((directory / "depth_config.json").read_text()) + return cls(model, tokenizer, depth_config) + + def _depth(self, effort): + if effort not in self.depths: + raise ValueError("effort must be low or high") + return self.depths[effort] + + def compile(self, request): + """Exact chat/no-thinking contract used in original native evaluation.""" + mapping = label_mapping(request) + prompt = self.tokenizer.apply_chat_template( + [{"role": "user", "content": render_prompt(request)}], + tokenize=False, + add_generation_prompt=True, + enable_thinking=False, + ) + ids = self.tokenizer.encode(prompt, add_special_tokens=False) + if not ids or len(ids) > self.max_length: + raise ValueError("input exceeds runtime limit; no truncation permitted") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.model.config, name, None) + if token is not None and token in ids: + raise ValueError("multimodal placeholders are unsupported") + tokens = [] + for label in mapping: + token = self.tokenizer.encode(label, add_special_tokens=False) + joint = self.tokenizer.encode(prompt + label, add_special_tokens=False) + if len(token) != 1 or joint != ids + token: + raise ValueError( + "candidate label is not single-token at answer boundary" + ) + if token[0] in self.tokenizer.all_special_ids: + raise ValueError("candidate label must not be special token") + tokens.append(token[0]) + if len(set(tokens)) != len(tokens): + raise ValueError("candidate token IDs must be unique") + return ids, tokens + + @torch.inference_mode() + def decide(self, request, effort="high"): + depth = self._depth(effort) + ids, candidates = self.compile(request) + input_ids = torch.tensor([ids], dtype=torch.long, device=self.device) + candidate_ids = torch.tensor(candidates, dtype=torch.long, device=self.device) + if effort == "high": + # Match the original reference high/full-vocabulary head path. + output = self.model( + input_ids=input_ids, + attention_mask=torch.ones_like(input_ids), + position_ids=torch.arange(len(ids), device=self.device).unsqueeze(0), + past_key_values=None, + use_cache=True, + return_dict=True, + logits_to_keep=1, + ) + logits = output.logits[0, -1].index_select(0, candidate_ids).float() + projection = "full_head" + else: + (decision,) = self.wrapper(input_ids, candidate_ids, depths=(depth,)) + logits = decision.logits[0].float() + projection = decision.projection_mode + response = format_response(request, logits.softmax(-1).tolist()) + response.update( + effort=effort, + executed_layers=depth, + prompt_tokens=len(ids), + logits=logits.tolist(), + projection=projection, + calibrated=False, + ) + return response + + @torch.inference_mode() + def generate_text(self, user_text, effort="high", max_new_tokens=128): + """TYPE greedy reference; replays prefix each token, not optimized serving.""" + depth = self._depth(effort) + if not isinstance(user_text, str) or not user_text.strip(): + raise ValueError("user_text must be nonempty") + if type(max_new_tokens) is not int or max_new_tokens < 1: + raise ValueError("max_new_tokens must be positive integer") + ids = self.tokenizer.apply_chat_template( + [{"role": "user", "content": user_text}], + tokenize=True, + add_generation_prompt=True, + enable_thinking=False, + return_dict=False, + ) + if not ids or len(ids) + max_new_tokens > self.max_length: + raise ValueError("prompt plus generation reservation exceeds limit") + eos = self.model.generation_config.eos_token_id + eos = [eos] if isinstance(eos, int) else list(eos or []) + generated = [] + for _ in range(max_new_tokens): + tensor = torch.tensor([ids + generated], device=self.device) + state = self.wrapper.advance(self.wrapper.begin(tensor), depth) + hidden = self.wrapper.backbone.norm(state.hidden[:, -1]) + token = self.model.get_output_embeddings()(hidden)[0].argmax().item() + generated.append(token) + if token in eos: + break + return { + "text": self.tokenizer.decode(generated, skip_special_tokens=True), + "token_ids": generated, + "effort": effort, + "executed_layers_per_token": depth, + "finish_reason": "eos" if generated[-1] in eos else "length", + "prompt_tokens": len(ids), + } diff --git a/9B-3000/processor_config.json b/9B-3000/processor_config.json new file mode 100644 index 0000000000000000000000000000000000000000..43c4343ec493f200e452f87a7c6649ecdfcce1ac --- /dev/null +++ b/9B-3000/processor_config.json @@ -0,0 +1,61 @@ +{ + "image_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_processor_type": "Qwen2VLImageProcessor", + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "merge_size": 2, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "size": { + "longest_edge": 16777216, + "shortest_edge": 65536 + }, + "temporal_patch_size": 2 + }, + "processor_class": "Qwen3VLProcessor", + "video_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "do_sample_frames": true, + "fps": 2, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "max_frames": 768, + "max_video_tokens": 768, + "merge_size": 2, + "min_frames": 4, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "return_metadata": false, + "size": { + "longest_edge": 25165824, + "shortest_edge": 4096 + }, + "temporal_patch_size": 2, + "video_processor_type": "Qwen3VLVideoProcessor" + } +} diff --git a/9B-3000/provenance/BASE-LICENSE.txt b/9B-3000/provenance/BASE-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/9B-3000/provenance/BASE-LICENSE.txt @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/9B-3000/provenance/SOURCE-PROJECT-LICENSE.txt b/9B-3000/provenance/SOURCE-PROJECT-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..261eeb9e9f8b2b4b0d119366dda99c6fd7d35c64 --- /dev/null +++ b/9B-3000/provenance/SOURCE-PROJECT-LICENSE.txt @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/9B-3000/provenance/code-license-provenance.json b/9B-3000/provenance/code-license-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..9d9d90c416a85a9210038a456ef1cb7d83253673 --- /dev/null +++ b/9B-3000/provenance/code-license-provenance.json @@ -0,0 +1,7 @@ +{ + "source_project": "xDAN-ms-swift-jev", + "source_paths_and_hashes": "source-provenance.json", + "scope": "Source provenance only; assembly does not choose or grant a new license for exported code, adapters or training data.", + "source_project_license_sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4", + "source_project_license_file": "SOURCE-PROJECT-LICENSE.txt" +} diff --git a/9B-3000/provenance/identity.json b/9B-3000/provenance/identity.json new file mode 100644 index 0000000000000000000000000000000000000000..2c4744609c27bea8e17c6be14c6d7ca35affe9dc --- /dev/null +++ b/9B-3000/provenance/identity.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "checkpoint_step": 3000, + "adapter_sha256": "14dd3cbaa26ced2ba5237dfff3aad93af4350dd9a43b869c7ce62cc9dd38d03b", + "adapter_config_sha256": "a3d03e9dfd4a3895d3a163ae679f931d957633deda178370ece7ebc86a1515ab", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/9B-3000/release-manifest.json b/9B-3000/release-manifest.json new file mode 100644 index 0000000000000000000000000000000000000000..368d4e243ffd93ea63f5c957e5efee77b1b6a6c7 --- /dev/null +++ b/9B-3000/release-manifest.json @@ -0,0 +1,208 @@ +{ + "format": "openjet.merged-release.v1", + "size": "9B", + "checkpoint_step": 3000, + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "numerical_comparison_passed": false, + "decision_parity_passed": true, + "decision_agreement": 160, + "artifact_reviewed_variant": false, + "assembly_stage": "final", + "weight_arithmetic_passed": true, + "runtime_validation_status": "passed", + "private_staging": true, + "fresh_hub_download_gpu_check": "not established by assembly; consult publication receipt", + "files": [ + { + "path": "LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "README.md", + "bytes": 1924, + "sha256": "8e6939c48caf1cd839c9c52c41ed7c766d79abcab255a697a8e859ea19d22f0d" + }, + { + "path": "RUNTIME.md", + "bytes": 5556, + "sha256": "d9397e883c8b8709297f394c7c65eb0ee6b17c0994902b6f6d518d924fd2c854" + }, + { + "path": "chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "config.json", + "bytes": 2832, + "sha256": "b1d02fb6b40aeba43105ca558b2532f31d062e0bd95bce564d9996c0a34b4211" + }, + { + "path": "decision-release.json", + "bytes": 1484, + "sha256": "01b56794252336f6a3385a06cd63688216039e125cb572ae6c0ca0675228a593" + }, + { + "path": "depth_config.json", + "bytes": 320, + "sha256": "f86116c627cd77eee275a9d1632b6783b085171ad85ade66699d266c432ab943" + }, + { + "path": "evaluation/runtime-smoke.json", + "bytes": 4905, + "sha256": "70d686828090965063966dddac6d8454cf981c6614a406c1647617986e534c5d" + }, + { + "path": "evaluation/weight-arithmetic.json", + "bytes": 25495, + "sha256": "efc2c717c9c8d62fec62ba9236db8f29166111961ade2f99d038b8c37aac54d4" + }, + { + "path": "examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "merge-provenance.json", + "bytes": 682, + "sha256": "3f6af915e561ce7fe67ec6ba64bf1e419923e23c609d71db4cf63aded0b202b3" + }, + { + "path": "merged-evaluation.json", + "bytes": 3395, + "sha256": "e0683fb077b2af2ca4193950a211c5e0a31560fb003b44c4c43e688248fa3587" + }, + { + "path": "model-00001-of-00006.safetensors", + "bytes": 2034237568, + "sha256": "dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344" + }, + { + "path": "model-00002-of-00006.safetensors", + "bytes": 3999615808, + "sha256": "3fe0ee3d29088f7f84ac7ba8c9d7562b91e6a31841c5b585c71779e618519e99" + }, + { + "path": "model-00003-of-00006.safetensors", + "bytes": 3997274128, + "sha256": "7bfce51b034a6de02c513b032c97007532676c5f915d020aa9a66f399f437117" + }, + { + "path": "model-00004-of-00006.safetensors", + "bytes": 3997290904, + "sha256": "f36d2b3ffc4c44dab06277ebbd722f73faca80aefc9aa3ce1119575f4a4773ae" + }, + { + "path": "model-00005-of-00006.safetensors", + "bytes": 3991239264, + "sha256": "b06602342e98eb882703857f3c8e8058894c03d4f2d62ddaab1e6dd4913c4e9b" + }, + { + "path": "model-00006-of-00006.safetensors", + "bytes": 800062816, + "sha256": "db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013" + }, + { + "path": "model.safetensors.index.json", + "bytes": 69253, + "sha256": "3d2f0ab780828a41449b9f32fee63e6ee0d27bf98fab0e52aa982f81e6c47cec" + }, + { + "path": "openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "provenance/identity.json", + "bytes": 682, + "sha256": "3f6af915e561ce7fe67ec6ba64bf1e419923e23c609d71db4cf63aded0b202b3" + }, + { + "path": "requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "runtime-validation.json", + "bytes": 334, + "sha256": "c291e6528feb11e87b878e6ea89676b2ef9fcb1cda7403efece382d891bb2755" + }, + { + "path": "source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "training.md", + "bytes": 2821, + "sha256": "0918aefaf0bf925ebf9aa7cbb9bbaa65ab5102de889ec0793b945f1936d84a0e" + } + ], + "family_documentation_update": { + "source_manifest_sha256": "b41961351cf7f6b20c156b320e9222feba636c711b9b9a1965eb3cba0c78de34", + "scope": "README only; source weights, runtime and evaluation evidence unchanged" + }, + "organization_documentation_update": { + "source_repo": "gump2049/APUS-OpenJev-v1", + "source_revision": "e7e3cc0b9c82b91380ec6595120b7ce8abd32fd7", + "previous_manifest_sha256": "0c94cf4789b07011b74aaac91ec6c7f6fa471959a2035e0f5ea141675588232c", + "scope": "README download links only; all model and runtime bytes unchanged" + } +} diff --git a/9B-3000/requirements.txt b/9B-3000/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..f015223d23acca43e86d5d1d8579ab7af82c2673 --- /dev/null +++ b/9B-3000/requirements.txt @@ -0,0 +1,5 @@ +# Install torch with the CUDA build matching the deployment host first. +torch==2.8.0 +transformers==5.16.1 +safetensors>=0.6 +huggingface_hub diff --git a/9B-3000/runtime-validation.json b/9B-3000/runtime-validation.json new file mode 100644 index 0000000000000000000000000000000000000000..8fcff74fdc494732f355e38391c5a9edd099d172 --- /dev/null +++ b/9B-3000/runtime-validation.json @@ -0,0 +1,10 @@ +{ + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "70d686828090965063966dddac6d8454cf981c6614a406c1647617986e534c5d" +} diff --git a/9B-3000/source-provenance.json b/9B-3000/source-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..e0432c1e4fb95b284334f882ed41af550fce5989 --- /dev/null +++ b/9B-3000/source-provenance.json @@ -0,0 +1,17 @@ +{ + "contracts.py": { + "source": "jev/dynamic/contracts.py", + "source_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "export_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + "candidate_projection.py": { + "source": "jev/dynamic/engine/candidate_projection.py", + "source_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6", + "export_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "early_exit.py": { + "source": "jev/dynamic/native/early_exit.py", + "source_sha256": "e6ce9e983df0fdaaa9fb9bdf4a6e1f37e0610708007648b2f6b0b9d36e35d2e3", + "export_sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + } +} diff --git a/9B-3000/tokenizer.json b/9B-3000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5520bfd2dd834ce386c1312c410fa71af56db5ad --- /dev/null +++ b/9B-3000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523 +size 19989325 diff --git a/9B-3000/tokenizer_config.json b/9B-3000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d134cd298be1e3be25db393d93a1cefe80e3214 --- /dev/null +++ b/9B-3000/tokenizer_config.json @@ -0,0 +1,33 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": true, + "local_files_only": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "processor_class": "Qwen3VLProcessor", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/9B-3000/training.md b/9B-3000/training.md new file mode 100644 index 0000000000000000000000000000000000000000..26f6df82b15c49cbdbc7b03049a2b0ad68eafebd --- /dev/null +++ b/9B-3000/training.md @@ -0,0 +1,24 @@ +# This release: 9B, step 3000 + +This is the intermediate 3000/5949 checkpoint. Curriculum totals below do not mean this checkpoint completed the whole curriculum. No 5949 checkpoint results are substituted. + +# Training provenance + +Both families use the registered 5949-record SFT curriculum (3898 parent groups), with 5949 maximum steps and one epoch. Each checkpoint's recorded training step, epoch, and original trainer-state hash are preserved in the separate [LoRA archive manifest](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/manifest.json) at fixed revision `1e5557f923746031f8187b7daf299b9bee41cb3c` (repository access required). This merged package's `release-manifest.json` inventories inference artifacts and does not contain that per-checkpoint trainer-state record; intermediate checkpoints did not finish the whole schedule. Randomized loader order means the step number alone is not a verified count of unique examples seen at an intermediate checkpoint. + +| Source | Scheduled records | Parent groups | +|---|---:|---:| +| Mind2Web browser Choice | 1798 | 671 | +| Mind2Web browser TYPE | 158 | 125 | +| HelpSteer3 principle | 1300 | 1300 | +| SGD | 513 | 10 | +| GoEmotions independent-attribute Score | 500 | 297 | +| BoolQ | 600 | 600 | +| MNLI | 1000 | 1000 | +| Local counterfactual | 80 | 20 | + +Parent groups can overlap across browser Choice/TYPE. The 4B run initialized from a 427-record pilot adapter (weight SHA256 `0047e5f1f0c98f17da93def94032041997592e1a609ddcab58de41f4d30e3a38`), while the inspected 9B config records no origin adapter. Do not add pilot records to the registered schedule as if all were independent. + +Decision objective: `0.5 CE(low) + 0.5 CE(high) + 0.1 KL(P_high.detach || P_low)`. TYPE examples use full-depth text cross-entropy. LoRA r=8, alpha=16, dropout=0; learning rate 1e-4; batch size 1; seed 20260920; training max length 6144. The recorded compiled schedule maximum is 5845 tokens. Training settings and data/schedule hashes are retained in the separate [LoRA checkpoint depth configuration](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/9b/checkpoint-3000/depth_config.json) at fixed archive revision `1e5557f923746031f8187b7daf299b9bee41cb3c`. The merged package's `depth_config.json` contains only portable inference and identity metadata; it is not the full training configuration. + +This is SFT with a within-model distillation term; it does not prove RLCD, online reinforcement learning or teacher-model OPD occurred. The declared public sources contain multiple licensing regimes (including CC-BY, CC-BY-SA and mixed-source material); separate provenance and redistribution review remains necessary. Raw datasets are not part of this upload. diff --git a/9B-5949/LICENSE b/9B-5949/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/9B-5949/LICENSE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/9B-5949/README.md b/9B-5949/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9216ada176dbe3f613cdeb8b56234d6e0934129e --- /dev/null +++ b/9B-5949/README.md @@ -0,0 +1,33 @@ +# APUS-OpenJev-v1 · 9B Research Variant + +A decision model for browser agents and business workflows. This directory contains standalone BF16 weights and a runtime with selectable `effort="low"` and `effort="high"` compute budgets. + +[Model family](../README.md) · [Architecture](../ARCHITECTURE.md) · [Runtime guide](RUNTIME.md) + +## Quick start + +Use a CUDA-capable PyTorch environment. + +```bash +python -m pip install huggingface_hub +hf auth login +hf download apus-ailab/APUS-OpenJev-v1 \ + --include "9B-5949/*" --local-dir ./APUS-OpenJev-v1 +cd ./APUS-OpenJev-v1/9B-5949 +python -m pip install -r requirements.txt +python examples.py . --device cuda:0 --effort high +``` + +The included runtime provides compute-budget selection. Use `high` for text generation. + +## Evaluation + +With the full compute budget, this merged model scores **67/80 (83.75%)** on the [Frozen80 development panel](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921): browser action selection, principle-based judgment, evidence-based questions, natural language inference, and attribute decisions. + +This reused development panel is an engineering reference, not an independent benchmark. BF16 merging changes some candidate probabilities; decision thresholds require revalidation. See [evaluation results](merged-evaluation.json) and [runtime checks](evaluation/runtime-smoke.json) for details. + +## Provenance + +We thank the Qwen team for the [Qwen3.5-9B](https://huggingface.co/Qwen/Qwen3.5-9B) base model. Training details and source records are in [training.md](training.md); artifact hashes are in [release-manifest.json](release-manifest.json). Consult [LICENSE](LICENSE) and the [base-model](provenance/BASE-LICENSE.txt) and [source-project](provenance/SOURCE-PROJECT-LICENSE.txt) notices. + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab) diff --git a/9B-5949/RUNTIME.md b/9B-5949/RUNTIME.md new file mode 100644 index 0000000000000000000000000000000000000000..b2b0a9dea897dcb39f1bca001a40d6be64749727 --- /dev/null +++ b/9B-5949/RUNTIME.md @@ -0,0 +1,55 @@ +# xDAN-openJet merged reference runtime + +本目录是可随 HF merged 仓库发布的完整 Python 源码闭包,不需要安装 ms-swift、PEFT 或原项目。发布选择为 **4B checkpoint-5949**、**9B checkpoint-3000** 与 **9B checkpoint-5949**;各目录必须保留自己的 `depth_config.json`、完整 Qwen config、tokenizer、chat template、generation config 和 merged safetensors。不得将不同 checkpoint 的权重或 shallow/full 结果拼在一起。 + +## 运行 + +使用 CUDA 对应 PyTorch 2.8.0 构建,安装 `requirements.txt`。运行依赖 Transformer 私有模型层接口,所以严格要求 `transformers==5.16.1`;版本升级需重做层级及数值验证。这是原生 PyTorch 单卡单请求参考实现,不是 vLLM 服务。 + +先把发布仓库的**固定 commit**完整下载到本机目录。仓库根目录有本目录中的 `openjet_runtime/` 与 `examples.py` 时: + +```bash +python -m pip install -r requirements.txt +python examples.py ./model-snapshot --device cuda:0 --effort both +python examples.py ./model-snapshot --device cuda:0 --effort high --text +``` + +若代码与权重同在下载快照根目录,进入快照后把 `./model-snapshot` 改为 `.`。 + +```python +from openjet_runtime import OpenJet +from examples import decision_examples + +model = OpenJet.from_pretrained("./model-snapshot") +two_candidates, sixteen_candidates = decision_examples() +print(model.decide(two_candidates, effort="low")) +print(model.decide(sixteen_candidates, effort="high")) +print(model.generate_text("Return only the text: red shoes", effort="high", max_new_tokens=32)) +``` + +`examples.py` 中两候选工作流、16 候选浏览器和 TYPE 是接口演示,不是声称模型已通过的 benchmark。需要针对业务设计 prompt 和验证答案。 + +## 接口和执行语义 + +- `decide(request, effort)` 输入字段与原 `jev.dynamic.prompt.v2` 相同:`id/group_id/state/instructions/primitive/criteria`;每个候选有非空 `id` 和 `description`,2–16 个,ID 不重复。标签为 A–P,编译器验证每个标签在真实回答边界恰为一个 token。没有 gold 输入需求。 +- `primitive="choice"` 返回 `choice` 与按输入顺序映射的 `probabilities`;`noul/score_level` 使用 `contracts.py` 中固定 Yes/No 候选,返回 `yes_probability`。`score_level` 是单个命题的判断,不能当作完整序数 Score API。 +- `effort="low"` 执行 `depth_config.exit_depth`(这两项发布预期16),共享原 LM final norm 与候选行投影;`high` 执行 `full_depth`(预期32),保持原评测中的标准模型前向+完整 LM head 路径。读取 config,不凭参数规模推测层数。 +- 所有输入采用 tokenizer 自带 chat template、`enable_thinking=False`,超过8192 token直接报错。不会默默截断 state、instructions 或候选。 +- `probabilities` 是在当前候选集上的相对 softmax,**未经概率校准**;合并不自动带来可信置信度或校准保证。 +- `generate_text` 为 TYPE 文本保留的贪心参考路径;每个 token 重新计算前缀,方便与原评测逐 token 核对,但不适合宣传 tokens/s。达到上限明确返回 `finish_reason="length"`。 +- `both` 示例分别调用两个 effort;没有自动路由、不承诺共享两次调用的前缀缓存。本次便携发布不包含 KV 广播引擎、vLLM 插件、TypeSafe HTTP server 或多模态输入能力。 + +## 合并验收(GPU,不能用静态测试代替) + +1. 固定同一 base revision、adapter SHA、tokenizer、chat template、dtype、attention backend 和 Transformers 版本,记录 merge 前后权重身份。保留 adapter 原文件。 +2. 同进程/新进程分别加载 base+adapter 和 merged;比较固定2候选、16候选、长输入、80题面板两 effort 的 token IDs、候选排序、logits、probabilities 和最终ID。保存逐题差异与最大绝对差,不能只比 aggregate accuracy。 +3. BF16 merge 会舍入:不能预先声明 bitwise一致或把漂移简单解释为无害;应报告数值误差和所有预测翻转。若超过事先制定的容忍值,停止发布数值等价结论,考虑 FP32 merge/存储再独立评测。 +4. fresh reload 验证 `depth_config` 与实际层数、模块边界一致。用层 forward hooks 检查 low只执行浅层、high执行全层;hooks会触发候选头保守fallback,不拿该测量做性能报告。 +5. TYPE短样本比较 token序列和EOS;分别测试空输入、非法候选/重复ID、超长输入明确失败。 +6. 在干净环境固定 HF revision 下载,运行本目录例子和同一小面板。记录显存、依赖、GPU型号以及权重checksum。成功加载只是第一关,不能当作质量或吞吐验收。 + +原始数值等价门结果:`False`,保留原结果;决策完全一致门:`False`,一致 `159/160`。独立运行时 GPU 验收状态:`passed`。若数值门失败,此 BF16 包作为独立重评版本,禁止直接迁移概率/拒答/路由阈值;见 `merged-evaluation.json`。 + +## 源码来历 + +`contracts.py`、`candidate_projection.py` 与 `early_exit.py` 从已有本地实现原样提取(最后一项仅调整相对 import);`source-provenance.json` 记录源路径与两端 SHA256。`runtime.py` 是最小加载及接口层;high 与 TYPE 分别对应原 `HFDecisionEngine._forward_batch` 和 `package_eval.prefix_next_token` 的执行语义。采用现有模型类:`qwen3_5` → `Qwen3_5ForConditionalGeneration`;`qwen3_5_text` → `AutoModelForCausalLM`。没有新增学习参数。 diff --git a/9B-5949/chat_template.jinja b/9B-5949/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..a585dec894e63da457d9440ec6aa7caa16d20860 --- /dev/null +++ b/9B-5949/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/9B-5949/config.json b/9B-5949/config.json new file mode 100644 index 0000000000000000000000000000000000000000..d9f290f1cb06a1e0e50ea151676447f0a59bd4a3 --- /dev/null +++ b/9B-5949/config.json @@ -0,0 +1,109 @@ +{ + "architectures": [ + "Qwen3_5ForConditionalGeneration" + ], + "dtype": "bfloat16", + "image_token_id": 248056, + "model_type": "qwen3_5", + "text_config": { + "attention_bias": false, + "attention_dropout": 0.0, + "attn_output_gate": true, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 248044, + "full_attention_interval": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 12288, + "layer_types": [ + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention" + ], + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 16, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "mamba_ssm_dtype": "float32", + "max_position_embeddings": 262144, + "mlp_only_layers": [], + "model_type": "qwen3_5_text", + "mtp_num_hidden_layers": 1, + "mtp_use_dedicated_embeddings": false, + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 4, + "pad_token_id": null, + "partial_rotary_factor": 0.25, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "mrope_interleaved": true, + "mrope_section": [ + 11, + 11, + 10 + ], + "partial_rotary_factor": 0.25, + "rope_theta": 10000000, + "rope_type": "default" + }, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 248320 + }, + "tie_word_embeddings": false, + "transformers_version": "5.16.1", + "video_token_id": 248057, + "vision_config": { + "deepstack_visual_indexes": [], + "depth": 27, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "in_channels": 3, + "initializer_range": 0.02, + "intermediate_size": 4304, + "model_type": "qwen3_5_vision", + "num_heads": 16, + "num_position_embeddings": 2304, + "out_hidden_size": 4096, + "patch_size": 16, + "spatial_merge_size": 2, + "temporal_patch_size": 2 + }, + "vision_end_token_id": 248054, + "vision_start_token_id": 248053 +} diff --git a/9B-5949/depth_config.json b/9B-5949/depth_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fb51c852597644abd381082d9d0d460b0b18e207 --- /dev/null +++ b/9B-5949/depth_config.json @@ -0,0 +1,10 @@ +{ + "prompt_version": "jev.dynamic.prompt.v2", + "exit_depth": 16, + "full_depth": 32, + "model_series": "xDAN-openJet", + "checkpoint_step": 5949, + "source_depth_config_sha256": "b6cab8011c151a7a5a5fce574addd8a5eb3d20e14cb516a1fb80d69563b4b4c9", + "training_mode": "two_exit", + "automatic_routing_validated": false +} diff --git a/9B-5949/evaluation/runtime-smoke.json b/9B-5949/evaluation/runtime-smoke.json new file mode 100644 index 0000000000000000000000000000000000000000..d32a699f414c0c25d4cdb626465ef229d50e5c70 --- /dev/null +++ b/9B-5949/evaluation/runtime-smoke.json @@ -0,0 +1,187 @@ +{ + "status": "passed", + "runtime_sha256": { + "openjet_runtime/__init__.py": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944", + "openjet_runtime/runtime.py": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c", + "openjet_runtime/early_exit.py": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566", + "openjet_runtime/contracts.py": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "openjet_runtime/candidate_projection.py": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "synthetic_examples": [ + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.9989758729934692, + "refund": 0.0010242064017802477 + }, + "choice": "close", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 103, + "logits": [ + 8.3125, + 1.4296875 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-binary", + "type": "choice", + "probabilities": { + "close": 0.999595582485199, + "refund": 0.00040448151412419975 + }, + "choice": "close", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 103, + "logits": [ + 20.75, + 12.9375 + ], + "projection": "full_head", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 0.008737827651202679, + "click-2": 0.007027901243418455, + "click-3": 0.05052885040640831, + "click-4": 0.019029337912797928, + "click-5": 0.007159519474953413, + "click-6": 0.0369676910340786, + "click-7": 0.013652882538735867, + "click-8": 0.029937351122498512, + "click-9": 0.02662683092057705, + "click-10": 0.022775137796998024, + "click-11": 0.04028511047363281, + "click-12": 0.1352221965789795, + "click-13": 0.08868088573217392, + "click-14": 0.19369687139987946, + "click-15": 0.03784435614943504, + "click-16": 0.28182727098464966 + }, + "choice": "click-16", + "effort": "low", + "executed_layers": 16, + "prompt_tokens": 383, + "logits": [ + 0.1201171875, + -0.09765625, + 1.875, + 0.8984375, + -0.0791015625, + 1.5625, + 0.56640625, + 1.3515625, + 1.234375, + 1.078125, + 1.6484375, + 2.859375, + 2.4375, + 3.21875, + 1.5859375, + 3.59375 + ], + "projection": "candidate_rows", + "calibrated": false + }, + { + "id": "example-browser", + "type": "choice", + "probabilities": { + "click-1": 3.5321256291354075e-05, + "click-2": 2.0125446098973043e-05, + "click-3": 1.1467132935649715e-05, + "click-4": 1.6684580259607174e-05, + "click-5": 1.1467132935649715e-05, + "click-6": 1.2206699466332793e-05, + "click-7": 8.38953292259248e-06, + "click-8": 2.1423424186650664e-05, + "click-9": 0.00019094489107374102, + "click-10": 2.75082238658797e-05, + "click-11": 0.0005190420779399574, + "click-12": 0.9989749193191528, + "click-13": 6.199072959134355e-05, + "click-14": 1.8906104742200114e-05, + "click-15": 1.1467132935649715e-05, + "click-16": 5.823490573675372e-05 + }, + "choice": "click-12", + "effort": "high", + "executed_layers": 32, + "prompt_tokens": 383, + "logits": [ + 11.125, + 10.5625, + 10.0, + 10.375, + 10.0, + 10.0625, + 9.6875, + 10.625, + 12.8125, + 10.875, + 13.8125, + 21.375, + 11.6875, + 10.5, + 10.0, + 11.625 + ], + "projection": "full_head", + "calibrated": false + } + ], + "text_smoke": { + "low": { + "text": "nesssss\u5730\u4e2d\u56fd\u5bb6\u5730...\u2026s\u6d0b\u65b0\u65b0\u53f8\u4ebatral", + "token_ids": [ + 248068, + 2022, + 82, + 753, + 95852, + 106947, + 95852, + 1076, + 1873, + 82, + 97648, + 95882, + 95882, + 95958, + 95765, + 175762 + ], + "effort": "low", + "executed_layers_per_token": 16, + "finish_reason": "length", + "prompt_tokens": 28 + }, + "high": { + "text": "red shoes\n", + "token_ids": [ + 1114, + 14850, + 248046, + 198, + 248044 + ], + "effort": "high", + "executed_layers_per_token": 32, + "finish_reason": "eos", + "prompt_tokens": 28 + } + }, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation" +} diff --git a/9B-5949/evaluation/weight-arithmetic.json b/9B-5949/evaluation/weight-arithmetic.json new file mode 100644 index 0000000000000000000000000000000000000000..87adc1c5fdcae41fb6b358f2b2b363a5adc1d90f --- /dev/null +++ b/9B-5949/evaluation/weight-arithmetic.json @@ -0,0 +1,908 @@ +{ + "status": "passed", + "base_mtp_keys_not_loaded_by_transformers": [ + "mtp.fc.weight", + "mtp.layers.0.input_layernorm.weight", + "mtp.layers.0.mlp.down_proj.weight", + "mtp.layers.0.mlp.gate_proj.weight", + "mtp.layers.0.mlp.up_proj.weight", + "mtp.layers.0.post_attention_layernorm.weight", + "mtp.layers.0.self_attn.k_norm.weight", + "mtp.layers.0.self_attn.k_proj.weight", + "mtp.layers.0.self_attn.o_proj.weight", + "mtp.layers.0.self_attn.q_norm.weight", + "mtp.layers.0.self_attn.q_proj.weight", + "mtp.layers.0.self_attn.v_proj.weight", + "mtp.norm.weight", + "mtp.pre_fc_norm_embedding.weight", + "mtp.pre_fc_norm_hidden.weight" + ], + "size": "9B-5949", + "targeted_weight_matrices": 176, + "adapter_tensors": 352, + "all_targeted_weights_exact": true, + "scope": "FP32 LoRA arithmetic followed by BF16 rounding; not forward/probability equivalence", + "adapter_sha256": "1d3a6af689bb841d7f5b70192ae32faec07489c4795a39de1399b0a16d64605e", + "checks": [ + { + "weight": "model.language_model.layers.0.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.0.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.1.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.10.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.11.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.12.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.13.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.14.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.15.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.16.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.17.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.18.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.19.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.2.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.20.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.21.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.22.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.23.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.24.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.25.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.26.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.27.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.28.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.29.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.3.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.30.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.31.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.4.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.5.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.6.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.k_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.o_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.q_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.7.self_attn.v_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.8.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.in_proj_qkv.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.linear_attn.out_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.down_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.gate_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + }, + { + "weight": "model.language_model.layers.9.mlp.up_proj.weight", + "equal": true, + "max_abs_difference": 0.0 + } + ] +} diff --git a/9B-5949/examples.py b/9B-5949/examples.py new file mode 100644 index 0000000000000000000000000000000000000000..c942c70fb5ee82412ae703b04eea3ea58d4226b8 --- /dev/null +++ b/9B-5949/examples.py @@ -0,0 +1,61 @@ +"""Run against a local merged HF snapshot; no project-local dependencies.""" + +import argparse +import json + +from openjet_runtime import OpenJet + + +def decision_examples(): + binary = { + "id": "example-binary", + "group_id": "example-binary", + "primitive": "choice", + "state": "Order 731 has been delivered. The customer's message says thank you.", + "instructions": "Select the appropriate next workflow action.", + "criteria": [ + {"id": "close", "description": "Close the resolved support ticket."}, + {"id": "refund", "description": "Refund an undelivered order."}, + ], + } + browser = { + "id": "example-browser", + "group_id": "example-browser", + "primitive": "choice", + "state": "A settings page has 16 visible buttons labeled Page 1 through Page 16.", + "instructions": "Navigate to Page 12 by choosing its matching button.", + "criteria": [ + {"id": f"click-{i}", "description": f"Click the Page {i} button."} + for i in range(1, 17) + ], + } + return binary, browser + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("model", help="Local merged snapshot directory") + parser.add_argument("--device", default="cuda:0") + parser.add_argument("--dtype", choices=["float32", "bfloat16"], default="bfloat16") + parser.add_argument("--effort", choices=["low", "high", "both"], default="both") + parser.add_argument( + "--text", action="store_true", help="Also run slow TYPE reference" + ) + args = parser.parse_args() + runtime = OpenJet.from_pretrained(args.model, args.device, args.dtype) + efforts = ("low", "high") if args.effort == "both" else (args.effort,) + for effort in efforts: + for request in decision_examples(): + result = runtime.decide(request, effort) + print(json.dumps({"example": request["id"], **result}, ensure_ascii=False)) + if args.text: + result = runtime.generate_text( + "Return only the literal text to type into a search box for 'red shoes'.", + effort=effort, + max_new_tokens=32, + ) + print(json.dumps({"example": "browser-type", **result}, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/9B-5949/generation_config.json b/9B-5949/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..f0b25ab7f068b3916a0b4e942812ee859e14c813 --- /dev/null +++ b/9B-5949/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "eos_token_id": 248044, + "transformers_version": "5.16.1", + "use_cache": true +} diff --git a/9B-5949/merge-provenance.json b/9B-5949/merge-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..fe5019f88186adc7be9f04106d6d5b13ac206d79 --- /dev/null +++ b/9B-5949/merge-provenance.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "checkpoint_step": 5949, + "adapter_sha256": "1d3a6af689bb841d7f5b70192ae32faec07489c4795a39de1399b0a16d64605e", + "adapter_config_sha256": "a3d03e9dfd4a3895d3a163ae679f931d957633deda178370ece7ebc86a1515ab", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/9B-5949/merged-evaluation.json b/9B-5949/merged-evaluation.json new file mode 100644 index 0000000000000000000000000000000000000000..326157fe33d1ce9f9ede724b612bd08f4da6ca00 --- /dev/null +++ b/9B-5949/merged-evaluation.json @@ -0,0 +1,81 @@ +{ + "scope": "80 fixed requests, both explicit efforts; merged-model comparison, not independent generalization evidence", + "checkpoint_step": 5949, + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "comparison": { + "passed": false, + "gate": "identical160_argmax_and_max_probability_delta_le_0.05", + "depths": { + "16": { + "before_correct": 66, + "after_correct": 67, + "changed_decisions": [ + { + "id": "mind2web:8368b990-c6ca-4cfe-a7ab-c2a88697639d:bf14f1d4-470f-4110-b3f4-019a9f7d0aed", + "before": "candidate-e76caeae9d53c8942ed1", + "after": "candidate-8b6280aa9682e1a79ce0", + "gold": "candidate-8b6280aa9682e1a79ce0" + } + ], + "max_probability_abs_difference": 0.07854090631008148, + "mean_probability_abs_difference": 0.0016820763134663653, + "max_logit_abs_difference": 0.375, + "mean_logit_abs_difference": 0.03666374206542969 + }, + "32": { + "before_correct": 67, + "after_correct": 67, + "changed_decisions": [], + "max_probability_abs_difference": 0.04566878080368042, + "mean_probability_abs_difference": 0.0006383866207283973, + "max_logit_abs_difference": 0.25, + "mean_logit_abs_difference": 0.03703125 + } + }, + "before_sha256": "4b26f109ac83f3833136c6fd158c59f5282d1752592fb71f8b27424ce18aabf6", + "after_sha256": "4ea7118b45a1b3fc94adef5bf2699210e87b1cfa2564e051b645d71c201278f7", + "verified_at_unix": 1789984732.9972832 + }, + "runtime_validation": { + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "1a9e55991d3a91dc4c704b080f6e251f3be244bfddb56246a0bdbcb00eb278b1" + }, + "decision_release": {}, + "reviewed_variant": { + "scope": "reviewed_variant_artifact", + "passed": true, + "private_only": true, + "post_observation_scope_amendment": true, + "original_comparison_sha256": "0a5ad3461c175596f1bcf76d0ef1142c2241b8d0044c1548d52a0e1f9cd0fb5d", + "allowed_changed_ids": [ + "mind2web:8368b990-c6ca-4cfe-a7ab-c2a88697639d:bf14f1d4-470f-4110-b3f4-019a9f7d0aed" + ], + "original_numerical_gate_passed": false, + "original_decision_parity_passed": false, + "decision_agreement": 159, + "total_decisions": 160, + "independent_questions": 80, + "no_calibration_transfer": true, + "probability_calibration_transfer_validated": false, + "review": "One low-depth Browser decision changed from wrong to correct; high unchanged. This authorizes an independently measured private BF16 variant, NOT an adapter-equivalent replacement and NOT an improvement claim. Both earlier equivalence gates remain failed.", + "before_top2_margin": 0.007057115435600281, + "after_top2_margin": 0.12943440675735474, + "required_final_gates": [ + "all adapted weight matrices equal expected merge arithmetic", + "portable runtime agrees with merged reference on160 decisions", + "all artifacts SHA verified after pinned HF download", + "fresh-process downloaded runtime verification" + ] + }, + "decision_parity_passed": false, + "decision_agreement": 159, + "weight_arithmetic_status": "passed", + "raw_inputs_included": false +} diff --git a/9B-5949/model-00001-of-00006.safetensors b/9B-5949/model-00001-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7a1670492c725c5be68f6ca47956025cbe4b1cf5 --- /dev/null +++ b/9B-5949/model-00001-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344 +size 2034237568 diff --git a/9B-5949/model-00002-of-00006.safetensors b/9B-5949/model-00002-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0e76d62a3e3957d6c5b8502856a92e1f7d3ed0f4 --- /dev/null +++ b/9B-5949/model-00002-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7dfbf5a64c3b383dfa996f7c438db87b6545453ad3092f1798005c0de4ad8299 +size 3999615808 diff --git a/9B-5949/model-00003-of-00006.safetensors b/9B-5949/model-00003-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..8771f49c8c411941778d79c27933319940f79273 --- /dev/null +++ b/9B-5949/model-00003-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1f2e7d9e1309ff2207a6004f4f749f6e7fc67ac303423e1a9c8eaaf381916d6 +size 3997274128 diff --git a/9B-5949/model-00004-of-00006.safetensors b/9B-5949/model-00004-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d3d7161ecccbfd2abd2cf201d3af9a2912b3175a --- /dev/null +++ b/9B-5949/model-00004-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db2c6e953a2899b1410b0bba753c1f29b618d5c5303d51c57300949f27f2a01e +size 3997290904 diff --git a/9B-5949/model-00005-of-00006.safetensors b/9B-5949/model-00005-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..bfa4e49c4e305b155cd4ebb6d81170a0974b0f0d --- /dev/null +++ b/9B-5949/model-00005-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cde34ae4e7a343a8ad96deee03516d0bdf3197bb6132727f47e8ada4a862b284 +size 3991239264 diff --git a/9B-5949/model-00006-of-00006.safetensors b/9B-5949/model-00006-of-00006.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..25da53d8687f76a008191a40e8d4c1829cdf221a --- /dev/null +++ b/9B-5949/model-00006-of-00006.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013 +size 800062816 diff --git a/9B-5949/model.safetensors.index.json b/9B-5949/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..5e44268776a44b9751d1aae62d822136b8ecb6ce --- /dev/null +++ b/9B-5949/model.safetensors.index.json @@ -0,0 +1,768 @@ +{ + "metadata": { + "total_parameters": 9409813744, + "total_size": 18819627488 + }, + "weight_map": { + "lm_head.weight": "model-00001-of-00006.safetensors", + "model.language_model.embed_tokens.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.0.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.1.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.10.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.10.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.k_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.q_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.11.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.12.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.13.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.13.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.14.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.k_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.q_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.15.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.16.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.17.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.18.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.k_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.k_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.o_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.q_norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.q_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.19.self_attn.v_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.2.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.2.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.20.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.20.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.21.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.A_log": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.conv1d.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.dt_bias": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_a.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_b.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_z.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.norm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.linear_attn.out_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.down_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.gate_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.mlp.up_proj.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.22.post_attention_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.23.input_layernorm.weight": "model-00004-of-00006.safetensors", + "model.language_model.layers.23.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.23.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.24.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.25.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.26.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.27.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.28.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.29.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.3.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.gate_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.mlp.up_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.post_attention_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.k_norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.k_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.o_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.q_norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.q_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.3.self_attn.v_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.30.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.A_log": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.conv1d.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.dt_bias": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_a.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_b.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_qkv.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_z.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.linear_attn.out_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.30.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.input_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.down_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.gate_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.mlp.up_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.post_attention_layernorm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.k_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.k_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.o_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.q_norm.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.q_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.31.self_attn.v_proj.weight": "model-00005-of-00006.safetensors", + "model.language_model.layers.4.input_layernorm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.A_log": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.conv1d.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.dt_bias": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_a.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_b.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_z.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.norm.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.linear_attn.out_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.mlp.down_proj.weight": "model-00002-of-00006.safetensors", + "model.language_model.layers.4.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.4.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.4.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.5.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.6.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.k_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.k_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.o_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.q_norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.q_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.7.self_attn.v_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.8.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.input_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.A_log": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.conv1d.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.dt_bias": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_a.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_b.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_z.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.norm.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.linear_attn.out_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.down_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.gate_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.mlp.up_proj.weight": "model-00003-of-00006.safetensors", + "model.language_model.layers.9.post_attention_layernorm.weight": "model-00003-of-00006.safetensors", + "model.language_model.norm.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.0.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.1.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.10.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.10.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.11.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.12.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.13.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.14.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.15.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.16.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.17.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.18.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.19.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.2.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.2.norm2.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.20.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.20.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.21.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.22.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.23.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.24.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.25.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.26.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.attn.proj.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.proj.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.qkv.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.attn.qkv.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.weight": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.bias": "model-00005-of-00006.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.3.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.4.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.5.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.6.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.7.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.8.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.qkv.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.attn.qkv.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm1.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm1.weight": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm2.bias": "model-00006-of-00006.safetensors", + "model.visual.blocks.9.norm2.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc1.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc1.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc2.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.linear_fc2.weight": "model-00006-of-00006.safetensors", + "model.visual.merger.norm.bias": "model-00006-of-00006.safetensors", + "model.visual.merger.norm.weight": "model-00006-of-00006.safetensors", + "model.visual.patch_embed.proj.bias": "model-00006-of-00006.safetensors", + "model.visual.patch_embed.proj.weight": "model-00006-of-00006.safetensors", + "model.visual.pos_embed.weight": "model-00006-of-00006.safetensors" + } +} diff --git a/9B-5949/openjet_runtime/__init__.py b/9B-5949/openjet_runtime/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..b314e3fca191cba379bb38ac6156c1f57a93ba93 --- /dev/null +++ b/9B-5949/openjet_runtime/__init__.py @@ -0,0 +1,3 @@ +from .runtime import OpenJet + +__all__ = ["OpenJet"] diff --git a/9B-5949/openjet_runtime/candidate_projection.py b/9B-5949/openjet_runtime/candidate_projection.py new file mode 100644 index 0000000000000000000000000000000000000000..7241c2a9985c91f726c2a145a52a09a90a3e6449 --- /dev/null +++ b/9B-5949/openjet_runtime/candidate_projection.py @@ -0,0 +1,108 @@ +"""Project selected native dense LM-head rows before matmul, preserving autograd. + +The caller must pass the actual output head, never an unwrapped adapter base_layer. +Unsupported heads raise; there is deliberately no automatic full-head fallback. +""" + +import torch +from torch import nn +from torch.nn import functional as F +from torch.nn.modules import module as module_hooks + + +class UnsupportedCandidateHead(ValueError): + """The head's semantics cannot be reproduced by plain selected-row linear.""" + + +def _check_head(head): + if type(head) is not nn.Linear: + raise UnsupportedCandidateHead("only exact torch.nn.Linear is supported") + if ( + head.forward.__func__ is not nn.Linear.forward + if hasattr(head.forward, "__func__") + else True + ): + raise UnsupportedCandidateHead("overridden forward is unsupported") + if head._modules or head._buffers or set(head._parameters) != {"weight", "bias"}: + raise UnsupportedCandidateHead( + "head contains extra modules, buffers or parameters" + ) + for name in ( + "_forward_hooks", + "_forward_pre_hooks", + "_backward_hooks", + "_backward_pre_hooks", + ): + if getattr(head, name, None) or getattr(module_hooks, "_global" + name, None): + raise UnsupportedCandidateHead("module hooks would be bypassed") + for value in (head.weight, head.bias): + if value is None: + continue + if ( + type(value) is not nn.Parameter + or value.is_quantized + or value.layout != torch.strided + or not value.is_floating_point() + or value.device.type == "meta" + ): + raise UnsupportedCandidateHead( + "requires ordinary dense floating-point Parameters" + ) + if head.weight is None or head.weight.shape != ( + head.out_features, + head.in_features, + ): + raise UnsupportedCandidateHead("invalid dense weight shape") + if head.bias is not None and ( + head.bias.shape != (head.out_features,) + or head.bias.dtype != head.weight.dtype + or head.bias.device != head.weight.device + ): + raise UnsupportedCandidateHead( + "bias shape, dtype or device does not match weight" + ) + + +def candidate_logits(head, hidden_states, token_ids): + """Return [..., K] logits in supplied token order, including repeated IDs. + + Dtype/autocast follow F.linear; no explicit precision conversion or detach. + Shape-dependent floating GEMM rounding may differ from full-vocabulary GEMM. + This validates indices, not tokenization, semantic labels, or probability mass. + """ + _check_head(head) + if type(hidden_states) not in (torch.Tensor, nn.Parameter): + raise TypeError("hidden_states must be an ordinary Tensor") + if ( + hidden_states.ndim < 1 + or hidden_states.shape[-1] != head.in_features + or not hidden_states.is_floating_point() + or hidden_states.layout != torch.strided + or hidden_states.device != head.weight.device + ): + raise ValueError( + "hidden_states shape, floating layout or device does not match head" + ) + if type(token_ids) is torch.Tensor: + if ( + token_ids.ndim != 1 + or token_ids.dtype != torch.long + or token_ids.device.type == "meta" + ): + raise ValueError("token_ids must be a one-dimensional int64 tensor") + indices = token_ids.to(device=head.weight.device) + elif isinstance(token_ids, (list, tuple)): + if any(type(value) is not int for value in token_ids): + raise TypeError("token IDs must be integers, not booleans or floats") + if any(value < 0 or value >= head.out_features for value in token_ids): + raise ValueError("token ID outside vocabulary") + indices = torch.tensor(token_ids, dtype=torch.long, device=head.weight.device) + else: + raise TypeError("token_ids must be a list, tuple or int64 tensor") + if not indices.numel(): + raise ValueError("at least one candidate token is required") + if bool(((indices < 0) | (indices >= head.out_features)).any()): + raise ValueError("token ID outside vocabulary") + weight = head.weight.index_select(0, indices) + bias = None if head.bias is None else head.bias.index_select(0, indices) + return F.linear(hidden_states, weight, bias) diff --git a/9B-5949/openjet_runtime/contracts.py b/9B-5949/openjet_runtime/contracts.py new file mode 100644 index 0000000000000000000000000000000000000000..8ce59a0564743435043ecc36a0577a16bfd42edc --- /dev/null +++ b/9B-5949/openjet_runtime/contracts.py @@ -0,0 +1,102 @@ +"""Small shared contract. Prompts use a strict whitelist of input fields.""" + +import json +import math + +PROMPT_VERSION = "jev.dynamic.prompt.v2" +LABELS = tuple("ABCDEFGHIJKLMNOP") +BINARY_CRITERIA = [ + {"id": "yes", "description": "The stated proposition is true."}, + {"id": "no", "description": "The stated proposition is false."}, +] + + +def validate_request(record): + for key in ("id", "group_id", "state", "instructions"): + if not isinstance(record.get(key), str) or not record[key].strip(): + raise ValueError(f"{key} must be a nonempty string") + if record.get("primitive") not in ("choice", "noul", "score_level"): + raise ValueError("unsupported primitive") + criteria = record.get("criteria") + if not isinstance(criteria, list) or not 2 <= len(criteria) <= len(LABELS): + raise ValueError("criteria must contain 2..16 candidates") + ids = [] + for candidate in criteria: + if not isinstance(candidate, dict): + raise TypeError("candidate must be an object") + for key in ("id", "description"): + if not isinstance(candidate.get(key), str) or not candidate[key].strip(): + raise ValueError(f"candidate {key} must be nonempty") + ids.append(candidate["id"]) + if len(set(ids)) != len(ids): + raise ValueError("duplicate candidate ids") + if record["primitive"] != "choice" and criteria != BINARY_CRITERIA: + raise ValueError("noul and score_level require canonical yes/no criteria") + + +def validate_record(record): + validate_request(record) + if record.get("gold") not in [c["id"] for c in record["criteria"]]: + raise ValueError("gold must be a candidate id") + if not isinstance(record.get("provenance"), dict): + raise TypeError("provenance must be an object") + + +def label_mapping(record): + validate_request(record) + return dict(zip(LABELS, (c["id"] for c in record["criteria"]))) + + +def render_prompt_parts(record): + """Text prefix/suffix; callers MUST check tokenizer boundary equivalence.""" + validate_request(record) + prefix = "Shared state:\n" + record["state"] + "\n\n" + task = { + "primitive": record["primitive"], + "instructions": record["instructions"], + "criteria": [ + {"label": label, "description": candidate["description"]} + for label, candidate in zip(LABELS, record["criteria"]) + ], + } + suffix = json.dumps(task, ensure_ascii=False, sort_keys=True) + suffix += ( + "\nReturn only the selected letter: " + + ", ".join(LABELS[: len(record["criteria"])]) + + ".\nAnswer:" + ) + return prefix, suffix + + +def render_prompt(record): + return "".join(render_prompt_parts(record)) + + +def to_messages(record): + validate_record(record) + inverse = {candidate: label for label, candidate in label_mapping(record).items()} + return { + "messages": [ + {"role": "user", "content": render_prompt(record)}, + {"role": "assistant", "content": inverse[record["gold"]]}, + ] + } + + +def format_response(record, probabilities): + """Map ordered candidate probabilities; score_level is NOT aggregate Score.""" + mapping = label_mapping(record) + values = list(probabilities) + if len(values) != len(mapping) or any( + not math.isfinite(p) or p < 0 or p > 1 for p in values + ): + raise ValueError("invalid probabilities") + if not math.isclose(sum(values), 1, abs_tol=1e-5): + raise ValueError("probabilities must sum to one") + distribution = dict(zip(mapping.values(), values)) + result = {"type": record["primitive"], "probabilities": distribution} + if record["primitive"] == "choice": + result["choice"] = max(distribution, key=distribution.get) + else: + result["yes_probability"] = distribution["yes"] + return result diff --git a/9B-5949/openjet_runtime/early_exit.py b/9B-5949/openjet_runtime/early_exit.py new file mode 100644 index 0000000000000000000000000000000000000000..ebe0e9f2d75fb73c0175ffe3494b7c768380e39d --- /dev/null +++ b/9B-5949/openjet_runtime/early_exit.py @@ -0,0 +1,209 @@ +"""Actual Q1 no-cache layer-prefix execution for native Qwen3.5 decisions. + +Mirrors the mask/position preparation of Transformers Qwen3_5TextModel (5.16.1). +This is a version-audited reference, not a generic model or cache implementation. +""" + +from dataclasses import dataclass, replace + +import torch +from torch import Tensor, nn +from transformers.masking_utils import ( + create_causal_mask, + create_recurrent_attention_mask, +) + +from .candidate_projection import ( + UnsupportedCandidateHead, + candidate_logits, +) + + +@dataclass(frozen=True) +class DepthContinuation: + hidden: Tensor # complete sequence residual, BEFORE final norm + position_ids: Tensor + position_embeddings: tuple[Tensor, Tensor] + masks: dict[str, Tensor | None] + depth: int + owner: object + model_signature: tuple + + +@dataclass +class DepthDecision: + depth: int + logits: Tensor # [1,C] + projection_mode: str + + +class QwenEarlyExit(nn.Module): + """Begin once, stop at a real depth, optionally continue without replay. + + Q1 means one unpadded complete input sequence. No cache or token generation. + Continuations are ephemeral: do not mutate parameters/train-mode between + begin/advance/readout, or persist them across optimizer steps. + """ + + def __init__(self, model: nn.Module): + super().__init__() + self.model = model + if self.base.config.model_type not in {"qwen3_5", "qwen3_5_text"}: + raise ValueError("only Qwen3.5 text/conditional models are supported") + if len(self.backbone.layers) != self.backbone.config.num_hidden_layers: + raise ValueError("layer count/config mismatch") + if not set(self.backbone.config.layer_types) <= { + "linear_attention", + "full_attention", + }: + raise ValueError("unsupported hybrid layer type") + self._owner = object() + + @property + def base(self): + return ( + self.model.get_base_model() + if hasattr(self.model, "get_base_model") + else self.model + ) + + @property + def backbone(self): + return ( + self.base.model.language_model + if self.base.config.model_type == "qwen3_5" + else self.base.model + ) + + @property + def full_depth(self): + return len(self.backbone.layers) + + def _signature(self): + # Reference guard: optimizer updates and mode/device changes invalidate + # all outstanding continuations. No .data mutation is supported. + return ( + tuple((id(module), module.training) for module in self.model.modules()), + tuple( + (id(parameter), parameter._version, parameter.device, parameter.dtype) + for parameter in self.model.parameters() + ), + ) + + def begin(self, input_ids: Tensor) -> DepthContinuation: + if ( + input_ids.dtype != torch.long + or input_ids.ndim != 2 + or input_ids.shape[0] != 1 + or input_ids.shape[1] < 1 + ): + raise ValueError("Q1 requires nonempty int64 input_ids [1,S], no padding") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.base.config, name, None) + if token is not None and (input_ids == token).any(): + raise ValueError("multimodal placeholders are unsupported") + embedding = self.base.get_input_embeddings() + if ((input_ids < 0) | (input_ids >= embedding.weight.shape[0])).any(): + raise ValueError("input token outside vocabulary") + hidden = embedding(input_ids) + positions = ( + torch.arange(input_ids.shape[1], device=hidden.device) + .view(1, 1, -1) + .expand(4, 1, -1) + ) + text_positions = positions[0] + kwargs = { + "config": self.backbone.config, + "inputs_embeds": hidden, + "attention_mask": None, + "past_key_values": None, + "position_ids": text_positions, + } + masks = { + "full_attention": create_causal_mask(**kwargs), + "linear_attention": create_recurrent_attention_mask(**kwargs), + } + rotary = self.backbone.rotary_emb(hidden, positions[1:]) + return DepthContinuation( + hidden, text_positions, rotary, masks, 0, self._owner, self._signature() + ) + + def _check_state(self, state): + if state.owner is not self._owner: + raise ValueError("continuation belongs to a different executor") + if state.model_signature != self._signature(): + raise ValueError( + "stale continuation: model parameters or training mode changed" + ) + + def advance(self, state: DepthContinuation, target_depth: int) -> DepthContinuation: + self._check_state(state) + if ( + type(target_depth) is not int + or not state.depth < target_depth <= self.full_depth + ): + raise ValueError("target depth must advance within the model") + hidden = state.hidden + for index in range(state.depth, target_depth): + hidden = self.backbone.layers[index]( + hidden, + position_embeddings=state.position_embeddings, + attention_mask=state.masks[self.backbone.config.layer_types[index]], + position_ids=state.position_ids, + past_key_values=None, + use_cache=False, + ) + return replace(state, hidden=hidden, depth=target_depth) + + def readout( + self, state: DepthContinuation, candidate_token_ids: Tensor + ) -> DepthDecision: + self._check_state(state) + if state.depth < 1: + raise ValueError("readout requires at least one executed layer") + head = self.base.get_output_embeddings() + if ( + candidate_token_ids.dtype != torch.long + or candidate_token_ids.ndim != 1 + or candidate_token_ids.numel() < 2 + ): + raise ValueError("require at least two int64 candidate tokens [C]") + if candidate_token_ids.device != state.hidden.device: + raise ValueError("candidate IDs must share hidden device") + if candidate_token_ids.unique().numel() != candidate_token_ids.numel(): + raise ValueError("duplicate candidate token") + if ( + (candidate_token_ids < 0) | (candidate_token_ids >= head.weight.shape[0]) + ).any(): + raise ValueError("candidate token outside vocabulary") + # Never replace the residual used by continuation with normalized hidden. + hidden = self.backbone.norm(state.hidden[:, -1]) + try: + logits = candidate_logits(head, hidden, candidate_token_ids) + mode = "candidate_rows" + except UnsupportedCandidateHead: + # Preserve adapters/hooks/parametrizations by executing the real head. + logits = head(hidden).index_select(-1, candidate_token_ids) + mode = "full_head_fallback" + return DepthDecision(state.depth, logits, mode) + + def forward( + self, input_ids: Tensor, candidate_token_ids: Tensor, *, depths: tuple[int, ...] + ): + if ( + not depths + or any(type(d) is not int or not 1 <= d <= self.full_depth for d in depths) + or list(depths) != sorted(set(depths)) + ): + raise ValueError("depths must be strictly increasing valid layer counts") + state = self.begin(input_ids) + decisions = [] + for depth in depths: + state = self.advance(state, depth) + decisions.append(self.readout(state, candidate_token_ids)) + return tuple(decisions) diff --git a/9B-5949/openjet_runtime/runtime.py b/9B-5949/openjet_runtime/runtime.py new file mode 100644 index 0000000000000000000000000000000000000000..09839883299e5d0d2359df5612ecc0f264366bdf --- /dev/null +++ b/9B-5949/openjet_runtime/runtime.py @@ -0,0 +1,176 @@ +"""Portable, single-request merged Qwen3.5 decision reference runtime.""" + +import json +from pathlib import Path + +import torch +import transformers + +from .contracts import PROMPT_VERSION, format_response, label_mapping, render_prompt +from .early_exit import QwenEarlyExit + + +class OpenJet: + """Explicit low/high, not automatic routing. Text-only; never truncates.""" + + def __init__(self, model, tokenizer, depth_config, max_length=8192): + if transformers.__version__ != "5.16.1": + raise RuntimeError("Layer execution is audited for transformers==5.16.1") + self.model = model.eval() + self.tokenizer = tokenizer + self.wrapper = QwenEarlyExit(self.model) + self.device = self.model.get_input_embeddings().weight.device + self.max_length = max_length + if type(max_length) is not int or not 1 <= max_length <= 8192: + raise ValueError("max_length must be an integer in 1..8192") + if depth_config.get("prompt_version") != PROMPT_VERSION: + raise ValueError("checkpoint prompt version does not match runtime") + full = depth_config.get("full_depth") + low = depth_config.get("exit_depth") + if full != self.wrapper.full_depth: + raise ValueError("checkpoint depth and model depth disagree") + if type(low) is not int or not 0 < low < full: + raise ValueError("checkpoint does not declare a trained shallow exit") + if self.wrapper.backbone.config.layer_types[low - 1] != "full_attention": + raise ValueError("shallow exit must be a full-attention boundary") + self.depths = {"low": low, "high": full} + + @classmethod + def from_pretrained(cls, directory, device="cuda:0", dtype="bfloat16"): + """Load a local HF snapshot (download explicitly with a pinned revision).""" + from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer + + directory = Path(directory) + if (directory / "adapter_config.json").exists(): + raise ValueError("Expected a merged snapshot, not an adapter directory") + if dtype not in ("float32", "bfloat16"): + raise ValueError("supported dtypes: float32, bfloat16") + config = AutoConfig.from_pretrained(directory, local_files_only=True) + loader = AutoModelForCausalLM + if config.model_type == "qwen3_5": + from transformers import Qwen3_5ForConditionalGeneration + + loader = Qwen3_5ForConditionalGeneration + elif config.model_type != "qwen3_5_text": + raise ValueError("Expected Qwen3.5 text or conditional-generation model") + model = loader.from_pretrained( + directory, + local_files_only=True, + dtype=getattr(torch, dtype), + attn_implementation="sdpa", + ).to(device) + tokenizer = AutoTokenizer.from_pretrained(directory, local_files_only=True) + depth_config = json.loads((directory / "depth_config.json").read_text()) + return cls(model, tokenizer, depth_config) + + def _depth(self, effort): + if effort not in self.depths: + raise ValueError("effort must be low or high") + return self.depths[effort] + + def compile(self, request): + """Exact chat/no-thinking contract used in original native evaluation.""" + mapping = label_mapping(request) + prompt = self.tokenizer.apply_chat_template( + [{"role": "user", "content": render_prompt(request)}], + tokenize=False, + add_generation_prompt=True, + enable_thinking=False, + ) + ids = self.tokenizer.encode(prompt, add_special_tokens=False) + if not ids or len(ids) > self.max_length: + raise ValueError("input exceeds runtime limit; no truncation permitted") + for name in ( + "image_token_id", + "video_token_id", + "vision_start_token_id", + "vision_end_token_id", + ): + token = getattr(self.model.config, name, None) + if token is not None and token in ids: + raise ValueError("multimodal placeholders are unsupported") + tokens = [] + for label in mapping: + token = self.tokenizer.encode(label, add_special_tokens=False) + joint = self.tokenizer.encode(prompt + label, add_special_tokens=False) + if len(token) != 1 or joint != ids + token: + raise ValueError( + "candidate label is not single-token at answer boundary" + ) + if token[0] in self.tokenizer.all_special_ids: + raise ValueError("candidate label must not be special token") + tokens.append(token[0]) + if len(set(tokens)) != len(tokens): + raise ValueError("candidate token IDs must be unique") + return ids, tokens + + @torch.inference_mode() + def decide(self, request, effort="high"): + depth = self._depth(effort) + ids, candidates = self.compile(request) + input_ids = torch.tensor([ids], dtype=torch.long, device=self.device) + candidate_ids = torch.tensor(candidates, dtype=torch.long, device=self.device) + if effort == "high": + # Match the original reference high/full-vocabulary head path. + output = self.model( + input_ids=input_ids, + attention_mask=torch.ones_like(input_ids), + position_ids=torch.arange(len(ids), device=self.device).unsqueeze(0), + past_key_values=None, + use_cache=True, + return_dict=True, + logits_to_keep=1, + ) + logits = output.logits[0, -1].index_select(0, candidate_ids).float() + projection = "full_head" + else: + (decision,) = self.wrapper(input_ids, candidate_ids, depths=(depth,)) + logits = decision.logits[0].float() + projection = decision.projection_mode + response = format_response(request, logits.softmax(-1).tolist()) + response.update( + effort=effort, + executed_layers=depth, + prompt_tokens=len(ids), + logits=logits.tolist(), + projection=projection, + calibrated=False, + ) + return response + + @torch.inference_mode() + def generate_text(self, user_text, effort="high", max_new_tokens=128): + """TYPE greedy reference; replays prefix each token, not optimized serving.""" + depth = self._depth(effort) + if not isinstance(user_text, str) or not user_text.strip(): + raise ValueError("user_text must be nonempty") + if type(max_new_tokens) is not int or max_new_tokens < 1: + raise ValueError("max_new_tokens must be positive integer") + ids = self.tokenizer.apply_chat_template( + [{"role": "user", "content": user_text}], + tokenize=True, + add_generation_prompt=True, + enable_thinking=False, + return_dict=False, + ) + if not ids or len(ids) + max_new_tokens > self.max_length: + raise ValueError("prompt plus generation reservation exceeds limit") + eos = self.model.generation_config.eos_token_id + eos = [eos] if isinstance(eos, int) else list(eos or []) + generated = [] + for _ in range(max_new_tokens): + tensor = torch.tensor([ids + generated], device=self.device) + state = self.wrapper.advance(self.wrapper.begin(tensor), depth) + hidden = self.wrapper.backbone.norm(state.hidden[:, -1]) + token = self.model.get_output_embeddings()(hidden)[0].argmax().item() + generated.append(token) + if token in eos: + break + return { + "text": self.tokenizer.decode(generated, skip_special_tokens=True), + "token_ids": generated, + "effort": effort, + "executed_layers_per_token": depth, + "finish_reason": "eos" if generated[-1] in eos else "length", + "prompt_tokens": len(ids), + } diff --git a/9B-5949/processor_config.json b/9B-5949/processor_config.json new file mode 100644 index 0000000000000000000000000000000000000000..43c4343ec493f200e452f87a7c6649ecdfcce1ac --- /dev/null +++ b/9B-5949/processor_config.json @@ -0,0 +1,61 @@ +{ + "image_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_processor_type": "Qwen2VLImageProcessor", + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "merge_size": 2, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "size": { + "longest_edge": 16777216, + "shortest_edge": 65536 + }, + "temporal_patch_size": 2 + }, + "processor_class": "Qwen3VLProcessor", + "video_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "do_sample_frames": true, + "fps": 2, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "max_frames": 768, + "max_video_tokens": 768, + "merge_size": 2, + "min_frames": 4, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "return_metadata": false, + "size": { + "longest_edge": 25165824, + "shortest_edge": 4096 + }, + "temporal_patch_size": 2, + "video_processor_type": "Qwen3VLVideoProcessor" + } +} diff --git a/9B-5949/provenance/BASE-LICENSE.txt b/9B-5949/provenance/BASE-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..f938136e3adacfd92be087f6e113b5d6d97f678f --- /dev/null +++ b/9B-5949/provenance/BASE-LICENSE.txt @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 Alibaba Cloud + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/9B-5949/provenance/SOURCE-PROJECT-LICENSE.txt b/9B-5949/provenance/SOURCE-PROJECT-LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..261eeb9e9f8b2b4b0d119366dda99c6fd7d35c64 --- /dev/null +++ b/9B-5949/provenance/SOURCE-PROJECT-LICENSE.txt @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/9B-5949/provenance/code-license-provenance.json b/9B-5949/provenance/code-license-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..9d9d90c416a85a9210038a456ef1cb7d83253673 --- /dev/null +++ b/9B-5949/provenance/code-license-provenance.json @@ -0,0 +1,7 @@ +{ + "source_project": "xDAN-ms-swift-jev", + "source_paths_and_hashes": "source-provenance.json", + "scope": "Source provenance only; assembly does not choose or grant a new license for exported code, adapters or training data.", + "source_project_license_sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4", + "source_project_license_file": "SOURCE-PROJECT-LICENSE.txt" +} diff --git a/9B-5949/provenance/identity.json b/9B-5949/provenance/identity.json new file mode 100644 index 0000000000000000000000000000000000000000..fe5019f88186adc7be9f04106d6d5b13ac206d79 --- /dev/null +++ b/9B-5949/provenance/identity.json @@ -0,0 +1,17 @@ +{ + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "checkpoint_step": 5949, + "adapter_sha256": "1d3a6af689bb841d7f5b70192ae32faec07489c4795a39de1399b0a16d64605e", + "adapter_config_sha256": "a3d03e9dfd4a3895d3a163ae679f931d957633deda178370ece7ebc86a1515ab", + "official80_sha256": "b3374e82f0e605762d40ab6455449c9cb2d315804a175d1cda985cba9beded35", + "software": { + "torch": "2.8.0+cu128", + "transformers": "5.16.1", + "peft": "0.20.0" + }, + "merge_arithmetic": "float32 CPU safe_merge then bfloat16 storage", + "inference_dtype": "bfloat16", + "attention": "sdpa", + "gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition" +} diff --git a/9B-5949/release-manifest.json b/9B-5949/release-manifest.json new file mode 100644 index 0000000000000000000000000000000000000000..4b84bcf4401af9b9b0cd6731c666b8f181848a77 --- /dev/null +++ b/9B-5949/release-manifest.json @@ -0,0 +1,208 @@ +{ + "format": "openjet.merged-release.v1", + "size": "9B-5949", + "checkpoint_step": 5949, + "base_id": "Qwen/Qwen3.5-9B", + "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "numerical_comparison_passed": false, + "decision_parity_passed": false, + "decision_agreement": 159, + "artifact_reviewed_variant": true, + "assembly_stage": "final", + "weight_arithmetic_passed": true, + "runtime_validation_status": "passed", + "private_staging": true, + "fresh_hub_download_gpu_check": "not established by assembly; consult publication receipt", + "files": [ + { + "path": "LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "README.md", + "bytes": 1941, + "sha256": "fa8d623c896606d578e3e8dbcad687ab9ef490811dfd9269c04b0570c008d546" + }, + { + "path": "RUNTIME.md", + "bytes": 5557, + "sha256": "5e353f905571c015d655d9f2e46cbcfab31028ad45c5261aa7bba4dcf59d7275" + }, + { + "path": "chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "config.json", + "bytes": 2832, + "sha256": "b1d02fb6b40aeba43105ca558b2532f31d062e0bd95bce564d9996c0a34b4211" + }, + { + "path": "depth_config.json", + "bytes": 320, + "sha256": "7729f1ed0408aeeff51718f765fd017b9e2e61fe1fd329b78abeb9cbb42a6129" + }, + { + "path": "evaluation/runtime-smoke.json", + "bytes": 4923, + "sha256": "1a9e55991d3a91dc4c704b080f6e251f3be244bfddb56246a0bdbcb00eb278b1" + }, + { + "path": "evaluation/weight-arithmetic.json", + "bytes": 25500, + "sha256": "3121372fff18946ad31f3d750f52d55c9f53516a0fb06aa525f4a9c3a94a1757" + }, + { + "path": "examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "merge-provenance.json", + "bytes": 682, + "sha256": "9e196f258b24eff17da41e08bc3ebf5914e3b2edad5196e9146f4d9286d6a0d9" + }, + { + "path": "merged-evaluation.json", + "bytes": 3446, + "sha256": "6bc22a9795815eb46f07f4c83e2ac20a861afbd7b1099e82b5f810ed24b4db9b" + }, + { + "path": "model-00001-of-00006.safetensors", + "bytes": 2034237568, + "sha256": "dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344" + }, + { + "path": "model-00002-of-00006.safetensors", + "bytes": 3999615808, + "sha256": "7dfbf5a64c3b383dfa996f7c438db87b6545453ad3092f1798005c0de4ad8299" + }, + { + "path": "model-00003-of-00006.safetensors", + "bytes": 3997274128, + "sha256": "f1f2e7d9e1309ff2207a6004f4f749f6e7fc67ac303423e1a9c8eaaf381916d6" + }, + { + "path": "model-00004-of-00006.safetensors", + "bytes": 3997290904, + "sha256": "db2c6e953a2899b1410b0bba753c1f29b618d5c5303d51c57300949f27f2a01e" + }, + { + "path": "model-00005-of-00006.safetensors", + "bytes": 3991239264, + "sha256": "cde34ae4e7a343a8ad96deee03516d0bdf3197bb6132727f47e8ada4a862b284" + }, + { + "path": "model-00006-of-00006.safetensors", + "bytes": 800062816, + "sha256": "db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013" + }, + { + "path": "model.safetensors.index.json", + "bytes": 69253, + "sha256": "3d2f0ab780828a41449b9f32fee63e6ee0d27bf98fab0e52aa982f81e6c47cec" + }, + { + "path": "openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "provenance/identity.json", + "bytes": 682, + "sha256": "9e196f258b24eff17da41e08bc3ebf5914e3b2edad5196e9146f4d9286d6a0d9" + }, + { + "path": "requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "reviewed-variant.json", + "bytes": 1254, + "sha256": "7c134fe8297dd19227aea63ba8cc5cade5ab6aa380569dcf020ab32b912b885a" + }, + { + "path": "runtime-validation.json", + "bytes": 334, + "sha256": "83b021ff809c981760e56667355e63c29806b909b96eebb1f876f4f63c06b150" + }, + { + "path": "source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "training.md", + "bytes": 2697, + "sha256": "c03d263c6f8b8abeac60cade36b5cde0667b75b5e26ddfdbfb8ab49361e2cdaf" + } + ], + "family_documentation_update": { + "source_manifest_sha256": "4c3d21b13ec8d10c05a7a3bfc44f772a698dfb0c6688a7365d951f7780cfe7b8", + "scope": "README only; source weights, runtime and evaluation evidence unchanged" + }, + "organization_documentation_update": { + "source_repo": "gump2049/APUS-OpenJev-v1", + "source_revision": "e7e3cc0b9c82b91380ec6595120b7ce8abd32fd7", + "previous_manifest_sha256": "4459bba4b117e9d2f886d41a57d96e0c9c2f9bd55ba3933bcaa45f16636ebf94", + "scope": "README download links only; all model and runtime bytes unchanged" + } +} diff --git a/9B-5949/requirements.txt b/9B-5949/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..f015223d23acca43e86d5d1d8579ab7af82c2673 --- /dev/null +++ b/9B-5949/requirements.txt @@ -0,0 +1,5 @@ +# Install torch with the CUDA build matching the deployment host first. +torch==2.8.0 +transformers==5.16.1 +safetensors>=0.6 +huggingface_hub diff --git a/9B-5949/reviewed-variant.json b/9B-5949/reviewed-variant.json new file mode 100644 index 0000000000000000000000000000000000000000..802a867cab9be82fd6dd895aa7955b5a2f4cab00 --- /dev/null +++ b/9B-5949/reviewed-variant.json @@ -0,0 +1,26 @@ +{ + "scope": "reviewed_variant_artifact", + "passed": true, + "private_only": true, + "post_observation_scope_amendment": true, + "original_comparison_sha256": "0a5ad3461c175596f1bcf76d0ef1142c2241b8d0044c1548d52a0e1f9cd0fb5d", + "allowed_changed_ids": [ + "mind2web:8368b990-c6ca-4cfe-a7ab-c2a88697639d:bf14f1d4-470f-4110-b3f4-019a9f7d0aed" + ], + "original_numerical_gate_passed": false, + "original_decision_parity_passed": false, + "decision_agreement": 159, + "total_decisions": 160, + "independent_questions": 80, + "no_calibration_transfer": true, + "probability_calibration_transfer_validated": false, + "review": "One low-depth Browser decision changed from wrong to correct; high unchanged. This authorizes an independently measured private BF16 variant, NOT an adapter-equivalent replacement and NOT an improvement claim. Both earlier equivalence gates remain failed.", + "before_top2_margin": 0.007057115435600281, + "after_top2_margin": 0.12943440675735474, + "required_final_gates": [ + "all adapted weight matrices equal expected merge arithmetic", + "portable runtime agrees with merged reference on160 decisions", + "all artifacts SHA verified after pinned HF download", + "fresh-process downloaded runtime verification" + ] +} diff --git a/9B-5949/runtime-validation.json b/9B-5949/runtime-validation.json new file mode 100644 index 0000000000000000000000000000000000000000..a39b453e3c80c85575464bc6f2099b26a8bf7b14 --- /dev/null +++ b/9B-5949/runtime-validation.json @@ -0,0 +1,10 @@ +{ + "status": "passed", + "decisions": 160, + "max_probability_delta": 0.0, + "no_jev_import": true, + "long_input_rejected": true, + "invalid_effort_rejected": true, + "text_scope": "Execution smoke only; not TYPE accuracy or speed validation", + "source_sha256": "1a9e55991d3a91dc4c704b080f6e251f3be244bfddb56246a0bdbcb00eb278b1" +} diff --git a/9B-5949/source-provenance.json b/9B-5949/source-provenance.json new file mode 100644 index 0000000000000000000000000000000000000000..e0432c1e4fb95b284334f882ed41af550fce5989 --- /dev/null +++ b/9B-5949/source-provenance.json @@ -0,0 +1,17 @@ +{ + "contracts.py": { + "source": "jev/dynamic/contracts.py", + "source_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a", + "export_sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + "candidate_projection.py": { + "source": "jev/dynamic/engine/candidate_projection.py", + "source_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6", + "export_sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + "early_exit.py": { + "source": "jev/dynamic/native/early_exit.py", + "source_sha256": "e6ce9e983df0fdaaa9fb9bdf4a6e1f37e0610708007648b2f6b0b9d36e35d2e3", + "export_sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + } +} diff --git a/9B-5949/tokenizer.json b/9B-5949/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5520bfd2dd834ce386c1312c410fa71af56db5ad --- /dev/null +++ b/9B-5949/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523 +size 19989325 diff --git a/9B-5949/tokenizer_config.json b/9B-5949/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d134cd298be1e3be25db393d93a1cefe80e3214 --- /dev/null +++ b/9B-5949/tokenizer_config.json @@ -0,0 +1,33 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": true, + "local_files_only": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "processor_class": "Qwen3VLProcessor", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/9B-5949/training.md b/9B-5949/training.md new file mode 100644 index 0000000000000000000000000000000000000000..030b9f5c8f75657702988de4732354272673bc64 --- /dev/null +++ b/9B-5949/training.md @@ -0,0 +1,24 @@ +# This release: 9B-5949, step 5949 + +This is the completed 5949-step SFT endpoint. + +# Training provenance + +Both families use the registered 5949-record SFT curriculum (3898 parent groups), with 5949 maximum steps and one epoch. Each checkpoint's recorded training step, epoch, and original trainer-state hash are preserved in the separate [LoRA archive manifest](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/manifest.json) at fixed revision `1e5557f923746031f8187b7daf299b9bee41cb3c` (repository access required). This merged package's `release-manifest.json` inventories inference artifacts and does not contain that per-checkpoint trainer-state record; intermediate checkpoints did not finish the whole schedule. Randomized loader order means the step number alone is not a verified count of unique examples seen at an intermediate checkpoint. + +| Source | Scheduled records | Parent groups | +|---|---:|---:| +| Mind2Web browser Choice | 1798 | 671 | +| Mind2Web browser TYPE | 158 | 125 | +| HelpSteer3 principle | 1300 | 1300 | +| SGD | 513 | 10 | +| GoEmotions independent-attribute Score | 500 | 297 | +| BoolQ | 600 | 600 | +| MNLI | 1000 | 1000 | +| Local counterfactual | 80 | 20 | + +Parent groups can overlap across browser Choice/TYPE. The 4B run initialized from a 427-record pilot adapter (weight SHA256 `0047e5f1f0c98f17da93def94032041997592e1a609ddcab58de41f4d30e3a38`), while the inspected 9B config records no origin adapter. Do not add pilot records to the registered schedule as if all were independent. + +Decision objective: `0.5 CE(low) + 0.5 CE(high) + 0.1 KL(P_high.detach || P_low)`. TYPE examples use full-depth text cross-entropy. LoRA r=8, alpha=16, dropout=0; learning rate 1e-4; batch size 1; seed 20260920; training max length 6144. The recorded compiled schedule maximum is 5845 tokens. Training settings and data/schedule hashes are retained in the separate [LoRA checkpoint depth configuration](https://huggingface.co/gump2049/xDAN-openJet-LoRA-Checkpoints/blob/1e5557f923746031f8187b7daf299b9bee41cb3c/9b/checkpoint-5949/depth_config.json) at fixed archive revision `1e5557f923746031f8187b7daf299b9bee41cb3c`. The merged package's `depth_config.json` contains only portable inference and identity metadata; it is not the full training configuration. + +This is SFT with a within-model distillation term; it does not prove RLCD, online reinforcement learning or teacher-model OPD occurred. The declared public sources contain multiple licensing regimes (including CC-BY, CC-BY-SA and mixed-source material); separate provenance and redistribution review remains necessary. Raw datasets are not part of this upload. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md new file mode 100644 index 0000000000000000000000000000000000000000..42fc65cc66388fadee2c7ab15cf5e0693e7e5e39 --- /dev/null +++ b/ARCHITECTURE.md @@ -0,0 +1,58 @@ +# Architecture: Decisions Across Compute Budgets + +[Technical Report (PDF)](TECHNICAL_REPORT.pdf) · [Full report (Markdown)](TECHNICAL_REPORT.md) + +APUS-OpenJev addresses a practical question: how can a language model make useful decisions without requiring the same amount of computation for every application? A workflow may need a choice among a few actions, while another task needs more extensive interpretation of the evidence. The design combines a shared language backbone, joint training across compute budgets, and an output path aligned with candidate decisions. + +## One shared semantic backbone + +The model retains Qwen's language understanding and existing network blocks. Each request supplies the task, evidence, and candidate descriptions in natural language. The candidates define what the application can do; the model learns to interpret their meaning in context. It does not require a new classifier with a permanent set of business labels for every workflow. + +The same backbone supports a shorter path and a full path. Both use the model's existing normalization and language-model output head to read decisions. This creates two ways to use the same learned representation, rather than maintaining separate models with unrelated decision rules. + +```mermaid +flowchart TD + T[Task, evidence, and dynamic candidates] --> B[Shared semantic backbone] + B --> S[Short-path decision] + B --> F[Full-path decision] + S -. supervised learning .-> L[Training: correctness and cross-depth consistency] + F -. supervision and guidance .-> L + S --> R[Inference: one caller-selected budget] + F --> R +``` + +The diagram shows the paths available during training and inference, not a requirement to execute both for every request. + +## Joint learning across compute budgets + +An early exit is useful only if the representation at that point is ready to support the task. Simply stopping a pretrained model sooner does not ensure that its intermediate features can produce a reliable decision. + +Training therefore supervises both paths on the same decision examples. The short path must learn to identify the correct candidate using the computation available to it. The full path also learns from the reference answer, preserving a direct objective for the larger budget. Shared trainable parameters connect these objectives: adaptation must support useful decisions at more than one point in the network. + +This is the central training constraint. The shorter path is a trained decision path, not an arbitrary cutoff. It can still be weaker on difficult tasks; training creates a usable tradeoff rather than eliminating that tradeoff. + +## Cross-depth decision consistency + +Correct-answer supervision provides the target choice, but the full path also expresses relative preferences across the other candidates. Those preferences provide an additional learning signal for the short path. + +A self-distillation objective encourages the short-path distribution to approach the full-path distribution for the same input. Gradients stop through the full-path target in this term; the full path continues learning through its own supervised objective. No external teacher is required for this mechanism. + +Consistency is encouraged, not guaranteed. The objective does not make the two budgets interchangeable or turn their probabilities into calibrated confidence. It gives the shorter path guidance about the decision structure learned with more computation. + +## Decision-aligned output and execution control + +At inference, the model scores short labels associated with the supplied candidates. Host code maps the result back to a candidate ID and assembles the response. This avoids generating the syntax of a structured answer token by token while keeping task interpretation inside the language model. + +Applications select `effort="low"` or `effort="high"` to balance compute cost and decision quality. Low uses the trained shorter path; high uses the full path. Each call follows its selected budget while using the same shared model parameters. + +Efficiency can come from executing fewer blocks and avoiding unnecessary output generation. Long-context input processing can still dominate cost, and these mechanisms do not establish a universal speedup. Matched end-to-end measurements are needed for deployment claims. + +## Scope and practical tradeoffs + +The design combines task conditioning, training constraints, and execution control within a shared decision model. It is intended for choices such as static browser actions, workflow routing, and evidence-based judgments. Applications still need useful candidates and task-specific evaluation. + +Open-ended TYPE text received full-path supervision; use high for text generation. Candidate scores are relative preferences, and merged weights require their own threshold validation. Quality results and timing boundaries are reported on the [main page](README.md). + +For reproducible details, see the model's [training notes](9B-3000/training.md), [runtime](9B-3000/openjet_runtime/runtime.py), and [merge evaluation](9B-3000/merged-evaluation.json). Each model directory preserves its own evidence. + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab) diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..d0c3fa9572ecf837a5f9b2996ba5fa29faf60e2e --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 xDAN2099, gumpcheng and APUS-OpenJev-v1 contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/LICENSE_NOTICES.md b/LICENSE_NOTICES.md new file mode 100644 index 0000000000000000000000000000000000000000..4326db8214ab9350060ca741821842fd4d817d79 --- /dev/null +++ b/LICENSE_NOTICES.md @@ -0,0 +1,14 @@ +# License scope + +The root [MIT License](LICENSE) applies to original APUS-OpenJev-v1 contributions by the named contributors, including the new model cards and charts. It does not replace licenses or copyright notices for third-party components. + +| Component | Applicable terms | +| --- | --- | +| Original release code and documentation | MIT, where newly authored and not otherwise marked | +| Qwen-derived model weights | Apache License 2.0 and the preserved base-model notices | +| Inherited runtime / source-project code | Existing Apache License 2.0 notices; no blanket relicensing | +| Evaluation data and training sources | Each source dataset's license; see the separate dataset repository | + +The model directories [4B](4B-5949/LICENSE), [9B](9B-3000/LICENSE), and [9B research variant](9B-5949/LICENSE) retain their original license files. Each also includes `provenance/BASE-LICENSE.txt` and `provenance/SOURCE-PROJECT-LICENSE.txt`. All existing notices remain intact. + +The MIT notice does not imply that all underlying weights, datasets, or third-party assets are MIT-licensed. Repository visibility and access permissions are separate from these license terms. diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..b19d755836391a7cb9f1fbf20a6be2499dc50b30 --- /dev/null +++ b/README.md @@ -0,0 +1,156 @@ +--- +base_model: +- Qwen/Qwen3.5-4B +- Qwen/Qwen3.5-9B +library_name: transformers +language: +- en +- zh +tags: +- decision-model +- dynamic-depth +- structured-output +--- + +# APUS-OpenJev-v1 + +

+ APUS Official Website + APUS AI Lab on Hugging Face + APUS AI Lab on GitHub + Original contributions: MIT +

+ +

Technical Report

+ +**Decision models with selectable compute depth.** + +[Model weights](https://huggingface.co/apus-ailab/APUS-OpenJev-v1/tree/main) · [Architecture](ARCHITECTURE.md) · [Evaluation data](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921) + +## 1. Introduction + +We introduce **APUS-OpenJev-v1**, a family of decision models for browser agents and business workflows. Given a task, context, and candidate actions, the model returns a scored choice that an application can execute. The family provides 4B and 9B variants with a common decision interface. + +**Architecture.** A shared language-model backbone supports multiple compute budgets. Task definitions and candidate meanings arrive through natural language, allowing the same model to handle different decision spaces. Applications select `effort="low"` or `effort="high"` through the included runtime. + +**Joint post-training.** Decision learning is coordinated across computation depths. Both execution paths learn from reference decisions, while the complete path supplies a distribution-level learning signal for the shorter path. This trains useful early decisions rather than relying on an untrained intermediate representation. The [architecture guide](ARCHITECTURE.md) explains the design and its trade-offs. + +**Decision-oriented inference.** The runtime scores candidate labels and maps them back to application actions. This removes the need to generate a structured answer token by token for bounded-choice tasks. Choose `effort="low"` or `effort="high"` to balance compute cost and decision quality. + +On our frozen 80-question development panel, **APUS-OpenJev 9B achieves 85.0% accuracy**, compared with **82.5% for the Jev API**. Its historical local reference implementation records **77.76 ms median model-forward latency**. Accuracy and latency are reported with their evaluation settings below. + +![Decision accuracy on the same 80-question panel: APUS 9B 85%, APUS 4B and Jev 82.5%, Laya complete-input 68.75%.](assets/accuracy.svg) + +*Figure 1. Same-panel decision accuracy. APUS results use the full compute budget; Laya uses its higher-scoring complete-input configuration after truncation was corrected.* + +## 2. Evaluation Results + +### Decision quality + +| Model | Correct / total | Accuracy | +| --- | ---: | ---: | +| **APUS-OpenJev 9B** | **68 /80** | **85.00%** | +| APUS-OpenJev 4B | 66 /80 | 82.50% | +| Jev API | 66 /80 | 82.50% | +| Laya · typed configuration, complete input | 55 /80 | 68.75% | + +All models are compared on the same frozen question identities and reference labels. Laya's complete-input typed configuration reproduces 55/80 across three runs; its complete-input English configuration scores 54/80. Repetition establishes consistency on these questions, not additional independent evidence. The two-question advantage over Jev is a panel result and does not establish broad or statistically significant superiority. + +### Results by task + +![Per-task decision accuracy for APUS-OpenJev 4B and 9B, the Jev API, and Laya with complete input.](assets/task-accuracy.svg) + +*Figure 2. Correct answers and accuracy for each task family. APUS models use the full compute budget; Laya uses the complete-input typed configuration. Each family contains 16 questions, so one answer changes accuracy by 6.25 percentage points.* + +The 4B model leads this panel's Browser subset, while 9B's gains come from principle-based judgments and natural language inference. Laya's corrected Browser result is 4/16. The task breakdown and source hashes are recorded in [subset-chart-data.json](subset-chart-data.json). + +### Response latency + +![P50 and P95 latency bars, with local model, local pipeline and remote API timing boundaries labeled.](assets/latency.svg) + +*Figure 3. Latency observations with explicit measurement boundaries. P50 and P95 use separately labeled linear scales. These measurements are not streaming first-token latency or text-generation throughput.* + +| Measured path | P50 | P95 | Timing boundary | +| --- | ---: | ---: | --- | +| APUS-OpenJev 9B reference | 77.76 ms | 505.16 ms | Local model forward | +| APUS-OpenJev 4B reference | 85.43 ms | 406.30 ms | Local model forward | +| Laya typed · complete input | 8.68 ms | 37.66 ms | Local `system_one`, including tokenization and output processing | +| Jev API | 436.96 ms | 4874.61 ms | Complete public HTTP response | + +APUS figures come from the historical RTX PRO 6000 reference implementation, not a new end-to-end test of this merged release. Laya's measured local path is faster; the displayed statistics are the medians of three per-run P50/P95 values. Public API times include network, queuing, and service overhead, so their ratio to a local forward time is not a model speedup. Exact values and aggregation are recorded in [chart-data.json](chart-data.json). + +## 3. Evaluation Set + +![Frozen80 composition: five task families with 16 questions each, 80 questions and 79 parent groups in total.](assets/dataset-subsets.svg) + +*Figure 4. Dataset composition, source datasets, and target capabilities. Parent groups identify related examples; equal question counts do not imply equal task difficulty or production traffic.* + +| Task family | Questions | Decision evaluated | +| --- | ---: | --- | +| Browser / Mind2Web | 16 | Select an action from a static page state | +| HelpSteer3 | 16 | Check a response against a supplied principle | +| BoolQ | 16 | Answer a binary question from evidence | +| MNLI | 16 | Distinguish entailment, neutral, and contradiction | +| Score / GoEmotions | 16 | Judge whether an individual attribute applies | + +The panel contains **80 questions from 79 parent groups** and is released as `validation`. It has informed development and model selection. All 16 Score labels are No, so an always-No strategy scores 100% on that subset; these results cannot establish positive-case Score performance. Browser evaluation covers offline action selection, not complete website tasks. + +The [dataset](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921) includes JSONL/Parquet, frozen identifiers, provenance, schema documentation, and verification code. It is a separate private repository with its own access permissions. + +## 4. Decision Encoding + +A request carries the task, supporting context, and 2–16 candidate descriptions. The runtime assigns request-local short labels, evaluates the legal candidate set, and returns the selected candidate ID with relative scores. Application code assembles the response. Candidate meanings can change between requests without adding a fixed business-category classifier. + +A legal output can still be the wrong decision. Candidate probabilities are not calibrated confidence, and business thresholds require validation. See the [runtime contract](9B-3000/RUNTIME.md) for the supported request format. + +## 5. Minimal Inference + +The repository is a model-family bundle. Choose a subdirectory containing complete BF16 weights and the reference runtime: + +| Variant | Model directory | Suggested use | +| --- | --- | --- | +| **9B** | `9B-3000/` | Quality-focused evaluation | +| **4B** | `4B-5949/` | Smaller parameter footprint | + +An additional 9B research variant is listed in the [artifact manifest](bundle-manifest.json). Load a model subdirectory, not the repository root. + +Use a CUDA-capable PyTorch environment. Select the model directory to download: + +```bash +python -m pip install huggingface_hub +hf auth login +hf download apus-ailab/APUS-OpenJev-v1 \ + --include "9B-3000/*" --local-dir ./APUS-OpenJev-v1 +cd ./APUS-OpenJev-v1/9B-3000 +python -m pip install -r requirements.txt +python examples.py . --device cuda:0 --effort high +``` + +For 4B, download `4B-5949/*` and enter that directory. The included runtime implements the budget control; ordinary Transformers loading does not enable it automatically. Text generation should use `high`. See [inference documentation](9B-3000/RUNTIME.md) and [examples](9B-3000/examples.py). + +## 6. Reproducing the Evaluation + +Use the fixed dataset revision `f15c828a1d926912f28bf5a8bf1e83f9c6b45c72`, preserve question and candidate order, and record the model revision, runtime, dtype, and compute budget. Keep reference labels out of the model input. Report decision accuracy separately from latency, and state precisely where timing starts and ends. + +The release records fixed source revisions and per-file hashes. Model files were downloaded and verified, with GPU checks on the source packages; this family bundle preserves those model bytes. Subsequent model-card changes do not represent new training or evaluation. Detailed merge results and diagnostic examples remain in the [9B evaluation records](9B-3000/merged-evaluation.json) and [4B evaluation records](4B-5949/merged-evaluation.json). BF16 merging changed some candidate probabilities, so calibration and routing thresholds must be revalidated. + +## 7. License + +Original APUS-OpenJev-v1 code and documentation contributed in this release are licensed under the [MIT License](LICENSE). Qwen-derived model weights and inherited code retain their applicable [Apache 2.0 license and notices](9B-3000/LICENSE); see the [license scope](LICENSE_NOTICES.md) for all variants. Dataset licenses are documented separately. + +## 8. Citation + +```bibtex +@misc{apusopenjev2026, + title = {APUS-OpenJev-v1: Decision Models with Selectable Compute Depth}, + author = {gumpcheng and zhangxu and {APUS AI-LAB}}, + year = {2026}, + url = {https://huggingface.co/apus-ailab/APUS-OpenJev-v1} +} +``` + +## 9. Contact + +Visit the [APUS official website](https://www.apusai.com) or [APUS AI Lab on Hugging Face](https://huggingface.co/apus-ailab). For model questions and feedback, open a discussion in this model repository's [Community tab](https://huggingface.co/apus-ailab/APUS-OpenJev-v1/discussions). + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab). diff --git a/README.zh-CN.md b/README.zh-CN.md new file mode 100644 index 0000000000000000000000000000000000000000..2a360e0a06dd8671e382694841c17ea2160eec17 --- /dev/null +++ b/README.zh-CN.md @@ -0,0 +1,158 @@ +--- +base_model: +- Qwen/Qwen3.5-4B +- Qwen/Qwen3.5-9B +library_name: transformers +language: +- en +- zh +tags: +- decision-model +- dynamic-depth +- structured-output +--- + +

+ APUS Official Website + APUS AI Lab on Hugging Face + APUS AI Lab on GitHub + Original contributions: MIT +

+ +

Technical Report

+ +[English](README.md) | 简体中文 + +# APUS-OpenJev-v1 + +**支持选择计算深度的决策模型。** + +[模型权重](https://huggingface.co/apus-ailab/APUS-OpenJev-v1/tree/main) · [架构介绍](ARCHITECTURE.md) · [评测数据](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921) + +## 1. 简介 + +**APUS-OpenJev-v1** 是一个决策模型系列,面向浏览器智能体和业务工作流。输入任务、上下文和候选动作后,模型返回带有评分的选择,供应用执行。该系列提供4B和9B版本,采用统一的决策接口。 + +**架构。** 同一个语言模型骨干支持多种计算预算。任务定义和候选项的含义通过自然语言输入,使同一模型能够处理不同的决策空间。应用通过附带的runtime选择`effort="low"`或`effort="high"`。 + +**联合后训练。** 不同计算深度下的决策学习共同进行。两条执行路径都从参考答案中学习,完整路径还通过候选概率分布为较短路径提供学习信号。这样,较早的决策出口经过任务训练,而不是直接依赖未经训练的中间表示。[架构指南](ARCHITECTURE.md)进一步介绍了这一设计及其取舍。 + +**面向决策的推理。** runtime对候选标签评分,再映射回应用动作。对于有界选择任务,无需逐token生成结构化答案。通过选择`effort="low"`或`effort="high"`,在计算成本与决策质量之间取得平衡。 + +在冻结的80题开发回归面板上,**APUS-OpenJev 9B的准确率为85.0%**,**Jev API为82.5%**。其历史本地参考实现测得的**模型前向计算中位延迟为77.76 ms**。下文分别说明准确率和延迟的评测条件。 + +![同一80题面板上的决策准确率:APUS 9B为85%,APUS 4B与Jev为82.5%,Laya完整输入配置为68.75%。](assets/accuracy.svg) + +*图1. 同题决策准确率。APUS结果采用完整计算预算;Laya采用修复截断后的完整输入配置中成绩较高的一项。这是重复使用的开发面板,不是独立的最终基准测试。* + +## 2. 评测结果 + +### 决策质量 + +| 模型 | 正确数 / 总数 | 准确率 | +| --- | ---: | ---: | +| **APUS-OpenJev 9B** | **68 /80** | **85.00%** | +| APUS-OpenJev 4B | 66 /80 | 82.50% | +| Jev API | 66 /80 | 82.50% | +| Laya · typed专项配置,完整输入 | 55 /80 | 68.75% | + +所有模型使用同一组冻结题目及参考标签。Laya的完整输入typed专项配置在三轮运行中均得到55/80,其完整输入英语配置为54/80。重复运行证明这些题目上的结果一致性,并不增加独立证据。比Jev多答对两题是当前面板上的结果,不能据此认定具有广泛优势或统计显著优势。 + +### 各任务表现 + +![APUS-OpenJev 4B、9B、Jev API及完整输入Laya的各任务准确率。](assets/task-accuracy.svg) + +*图2:每类任务的正确题数与准确率。APUS采用完整计算预算,Laya采用完整输入的专项配置。每类16题,答对一题就改变6.25个百分点。* + +4B在本面板的Browser子集上得分最高,9B的增益来自原则判断和自然语言推断。Laya修复截断后的Browser成绩为4/16。分项数据与来源哈希见[subset-chart-data.json](subset-chart-data.json)。 + +### 响应延迟 + +![P50与P95延迟柱状图,分别标注本地模型、本地处理流程和远程API的计时边界。](assets/latency.svg) + +*图3. 明确标注测量边界的延迟观测。P50与P95使用分别标注的线性刻度。这些测量不是流式首token延迟,也不是文本生成吞吐量。* + +| 测量路径 | P50 | P95 | 计时边界 | +| --- | ---: | ---: | --- | +| APUS-OpenJev 9B参考实现 | 77.76 ms | 505.16 ms | 本地模型前向计算 | +| APUS-OpenJev 4B参考实现 | 85.43 ms | 406.30 ms | 本地模型前向计算 | +| Laya typed · 完整输入 | 8.68 ms | 37.66 ms | 本地`system_one`,包含分词和输出处理 | +| Jev API | 436.96 ms | 4874.61 ms | 公网完整HTTP响应 | + +APUS数字来自历史RTX PRO 6000参考实现,不是对本次合并发布模型进行的新一轮端到端测试。Laya已测的本地路径更快;表中数字取三轮各自P50/P95的中位数。公网API时间包含网络、排队和服务开销,因此不能把它与本地前向耗时相除,作为模型加速倍数。精确数据及聚合方式记录在[chart-data.json](chart-data.json)中。 + +## 3. 评测集 + +![Frozen80数据构成:五类任务各16题,共80题、79个父组。](assets/dataset-subsets.svg) + +*图4:数据子集、来源数据集及目标能力。父组用于识别相关样本;相同题量不代表难度相同,也不代表实际业务流量占比。* + +| 任务类型 | 题数 | 评估的决策能力 | +| --- | ---: | --- | +| Browser / Mind2Web | 16 | 根据静态页面状态选择动作 | +| HelpSteer3 | 16 | 按给定原则检查回答 | +| BoolQ | 16 | 根据证据回答二元问题 | +| MNLI | 16 | 区分蕴含、中立和矛盾关系 | +| Score / GoEmotions | 16 | 判断某个具体属性是否成立 | + +该面板包含**来自79个父样本组的80道题**,以`validation`用途发布,已用于研发和模型选择。全部16个Score标签均为No,因此始终选择No即可在该子集取得100%;这些结果不能证明Score正例判断能力。Browser评测覆盖离线动作选择,不代表完整网站任务的完成能力。 + +[数据集](https://huggingface.co/datasets/gump2049/xDAN-openJet-Eval-Frozen80-20260921)包含JSONL/Parquet、冻结标识、来源记录、数据结构说明及验证代码。它是独立的私有仓库,需要单独的访问权限。 + +## 4. 决策编码 + +每个请求提供任务、支撑上下文和2–16个候选描述。runtime为当前请求分配短标签,在合法候选集合上评分,并返回选中的候选ID及相对分数。应用代码负责组装响应。候选含义可以随请求变化,无需新增一个固定业务类别的分类器。 + +合法输出仍可能是错误决策。候选概率不是经过校准的置信度,业务阈值需要验证。支持的请求格式见[runtime接口约定](9B-3000/RUNTIME.md)。 + +## 5. 最小推理示例 + +本仓库集中保存一个模型系列。请选择包含完整BF16权重和参考runtime的具体子目录: + +| 版本 | 模型目录 | 建议用途 | +| --- | --- | --- | +| **9B** | `9B-3000/` | 优先关注质量的评测 | +| **4B** | `4B-5949/` | 较小的参数规模 | + +另一项9B研究版本列在[产物清单](bundle-manifest.json)中。请加载模型子目录,不要加载仓库根目录。 + +使用支持CUDA的PyTorch环境。以下示例下载指定模型目录: + +```bash +python -m pip install huggingface_hub +hf auth login +hf download apus-ailab/APUS-OpenJev-v1 \ + --include "9B-3000/*" --local-dir ./APUS-OpenJev-v1 +cd ./APUS-OpenJev-v1/9B-3000 +python -m pip install -r requirements.txt +python examples.py . --device cuda:0 --effort high +``` + +使用4B时,下载`4B-5949/*`并进入该目录。预算控制由附带的runtime实现,普通Transformers加载不会自动启用。文本生成应使用`high`。详见[推理文档](9B-3000/RUNTIME.md)及[示例](9B-3000/examples.py)。 + +## 6. 复现评测 + +使用固定的数据集版本`f15c828a1d926912f28bf5a8bf1e83f9c6b45c72`,保留题目与候选顺序,记录模型版本、runtime、dtype及计算预算。参考标签不得进入模型输入。准确率和延迟应分别报告,并准确说明计时起点与终点。 + +发布产物记录了固定来源版本和逐文件哈希。模型文件已完成下载与核验,源模型包经过GPU检查;本系列仓库保留了相同的模型字节。后续模型卡更新不代表新增训练或评测。详细合并结果和诊断示例保留在[9B评测记录](9B-3000/merged-evaluation.json)和[4B评测记录](4B-5949/merged-evaluation.json)中。BF16合并改变了部分候选概率,因此校准与路由阈值必须重新验证。 + +## 7. 许可 + +本次发布贡献的 APUS-OpenJev-v1 原创代码和文档采用 [MIT License](LICENSE)。衍生自 Qwen 的模型权重及继承代码保留适用的 [Apache 2.0 许可与声明](9B-3000/LICENSE);各版本的具体范围见[许可说明](LICENSE_NOTICES.md)。数据集许可单独说明。 + +## 8. 引用 + +```bibtex +@misc{apusopenjev2026, + title = {APUS-OpenJev-v1: Decision Models with Selectable Compute Depth}, + author = {gumpcheng and zhangxu and {APUS AI-LAB}}, + year = {2026}, + url = {https://huggingface.co/apus-ailab/APUS-OpenJev-v1} +} +``` + +## 9. 联系 + +欢迎访问 [APUS 官方网站](https://www.apusai.com)或 [APUS AI Lab 的 Hugging Face 主页](https://huggingface.co/apus-ailab)。模型使用问题和反馈可在本模型仓库的 [Community 页面](https://huggingface.co/apus-ailab/APUS-OpenJev-v1/discussions)发起讨论。 + +**Authors:** gumpcheng ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)), zhangxu, [APUS AI-LAB](https://github.com/APUS-AI-Lab)。 diff --git a/TECHNICAL_REPORT.md b/TECHNICAL_REPORT.md new file mode 100644 index 0000000000000000000000000000000000000000..2f138fec2929e9bb42214b05f637e6df4ce340db --- /dev/null +++ b/TECHNICAL_REPORT.md @@ -0,0 +1,176 @@ +# APUS-OpenJev-v1: Learning to Decide Across Compute Budgets + +**Technical Report v1.1** + +**gumpcheng** ([https://huggingface.co/xDAN2099](https://huggingface.co/xDAN2099)) · **zhangxu** +**[APUS AI-LAB](https://github.com/APUS-AI-Lab)** + +## Abstract + +Many useful language-model interactions end in a decision: choosing a browser action, applying a principle, routing a workflow, or determining whether evidence supports an attribute. These interactions require language understanding, but their outputs often occupy a small, explicitly defined space. APUS-OpenJev-v1 organizes model computation around that distinction. It combines language-conditioned candidate decisions, jointly trained execution depths, and a text objective for workflow steps that require open responses. A shared model exposes low and high effort modes, connecting application budgets to the depth of neural computation. Cross-depth self-distillation transfers information from the deeper decision distribution into an earlier exit. We describe the delivered training framework, its 4B and 9B instantiations, and Jet-DCRL, a separately implemented research extension for decision utility and probability quality. Development results establish an initial accuracy reference; the evaluation agenda targets the broader relationship between decision quality, execution cost, and workflow outcomes. + +### Model value + +**APUS-OpenJev brings language understanding to the decision points where applications need an actionable choice.** Its value comes from combining decision quality, a direct output path, and control over computational effort within one deployable model family. + +- **Evidence-informed decisions.** Tasks, policies, and candidate actions are expressed in natural language, so a shared model can address changing business rules and action spaces. The 9B model achieves 85.00% accuracy on the development comparison panel, alongside 82.50% for the Jev API. Section 7 presents the shared evaluation setting and results. +- **Less work between understanding and action.** Candidate scoring produces a bounded decision without decoding a complete structured document. This targets the repeated selection steps found in browser agents, business routing, and workflow handoffs. +- **Compute that fits the workflow.** Jointly trained low and high effort modes let applications choose the execution budget appropriate to each decision point. This creates a practical basis for balancing decision quality with latency and resource cost. +- **Deployment under application control.** Downloadable 4B and 9B weights and an accompanying runtime support local decision execution, giving teams control over infrastructure, data handling, and integration into existing systems. + +For browser agents, the target is choosing the next action from page evidence and a goal. For high-frequency business services, it is applying a supplied rule or routing a request to an appropriate next step. For workflow handoffs, it is returning a decision that host code can map into a structured action. These scenarios share the same language-conditioned decision interface; their production value can be evaluated through both decision correctness and completed workflow outcomes. + +## 1. From Text Completion to Language-Conditioned Decisions + +The value of a decision model lies in interpreting a changing problem, rather than memorizing a fixed business taxonomy. A browser page presents different elements at every step. A workflow may introduce a new policy, tool, or routing destination. The relevant candidate set therefore belongs to the request itself. APUS-OpenJev-v1 receives the context, question or instruction, and candidate descriptions in language, then returns a decision over that request-specific set. + +This representation preserves a central strength of language models: task meaning is communicated through the input. A common model can process evidence-based questions, natural-language principles, workflow states, and browser actions without requiring a separate output classifier for every business category. Compact output labels identify candidates; their meaning comes from the descriptions supplied for the current request. They are an interface to the decision space, not permanent class identities. + +For candidates `A(x) = {a₁, …, aₖ}`, an execution depth `d` produces logits `z_d(x, A)`. The decision distribution is: + +`p_d(aᵢ | x, A) = exp(z_d,i) / Σⱼ exp(z_d,j)`. + +Normalization takes place within the legal candidate set. A deterministic application layer can map the selected label into its structured response. This removes the need to generate JSON punctuation or repeatedly decode a known answer schema. It guarantees membership in the supplied candidate set, while semantic correctness remains a learned capability that must be evaluated. Open text is retained where the task genuinely requires it, such as entering a value during a browser workflow. + +## 2. A Shared Model Across Compute Budgets + +APUS-OpenJev-v1 makes execution depth an explicit part of the decision framework. The same parameter set supports an earlier decision exit and a deeper exit. Applications select `effort="low"` or `effort="high"` according to the latency and quality needs of a workflow. The two modes change the neural computation performed, rather than merely changing a response-length limit. + +
+Compute-depth architecture +
Figure 1. Shared language representations support decisions at different execution depths. Joint training connects the exits through supervision and distributional knowledge transfer; deployment exposes the resulting compute-budget choices.
+
+ +The shared prefix builds a contextual representation of the instruction, evidence, and candidate semantics. An earlier readout extracts a candidate decision from that representation. The deeper path applies additional backbone transformations before producing its decision. Both exits use the model's language output space, preserving a common interpretation of candidate labels across depths. This avoids introducing a collection of task-specific business heads as the organizing architecture. + +Depth-sensitive execution also requires a careful separation between the representation used for a decision and the state used for further computation. In the depth executor, normalization for a readout does not overwrite the residual state required by subsequent blocks. This distinction makes a shallow decision and a deeper continuation conceptually compatible. The current release establishes the selectable exit mechanisms; learned per-request routing and production continuation reuse remain separate evaluation targets. + +The important training question is whether an intermediate representation has acquired enough decision structure to be useful. Merely exposing an earlier layer does not answer it. Our approach therefore supervises the exits jointly and couples their candidate distributions. Lower-cost execution becomes a training objective with measurable quality, rather than an inference shortcut assumed to preserve behavior. + +## 3. Learning Decisions and Execution Depth Together + +### 3.1 Joint supervision and cross-depth knowledge transfer + +For a supervised decision with target distribution `q`, the delivered objective is: + +`L_decision = 0.5 CE(q, p_low) + 0.5 CE(q, p_high)` + +` + 0.1 KL(stopgrad(p_high) || p_low)`. + +Both exits receive direct task supervision. The deeper exit additionally acts as a distributional teacher for the earlier exit. Its detached probabilities convey relative candidate preferences beyond the identity of the correct answer, while stop-gradient keeps the distillation target from being changed through that term. The deeper pathway continues to learn through its own supervised objective. + +This arrangement is an internal transfer mechanism: the model learns to express useful decision structure at more than one computational budget. It is especially relevant when several candidates are plausible and the earlier exit must preserve their distinctions with fewer transformations. Its effect can be measured through the quality–cost curve of the two exits. + +The supervision and transfer terms operate together during each decision update, enabling the shared model to learn both execution budgets in one coordinated optimization process. + +### 3.2 Text co-training for workflow continuity + +Some decisions lead directly to an open response: a search query, a form value, or another textual argument. TYPE examples therefore use the full-depth native text cross-entropy objective. Decision and text examples share the model while following the objective appropriate to their output space. + +The resulting training interface separates bounded selection from open completion without discarding either capability. A browser-oriented system can learn which action to take and retain a path for supplying the text that action requires. This is a training design for workflow continuity; full end-to-end browser competence requires additional execution-based evaluation beyond static action selection. + +### 3.3 A staged 4B path and an independent 9B study + +The 4B model follows a concrete staged adaptation path. A full-depth starter first learns from a focused mixture of decision and TYPE examples. The resulting parameters initialize a broader run that jointly optimizes the two decision exits and the text objective. The initial stage establishes task-facing behavior; the subsequent stage introduces the shared compute-budget learning objective across a wider task mixture. + +The 9B study starts independently from its own base initialization and uses the same registered curriculum and joint objective. It does not inherit the 4B starter. This provides a model-scale comparison with a common main training protocol while preserving the distinction in initialization history. The primary 9B result reported here uses an intermediate training version; the full-schedule version is also retained. + +
+Learning framework +
Figure 2. The delivered framework combines task alignment, mixed decision/text supervision, and joint depth learning. Jet-DCRL extends the research program toward candidate utility and probability quality; it is not part of the published weights described here.
+
+ +## 4. A Curriculum Organized Around Decision Capabilities + +The main curriculum allocates **97.34%** of training records to candidate decisions and **2.66%** to open-text TYPE examples. The source shares below describe the registered training mixture. Related examples are tracked by parent episode, dialogue, or source context to preserve their provenance. + +| Training source | Share | Capability emphasized | +|---|---:|---| +| Mind2Web browser Choice | 30.22% | Selecting actions from page evidence and goals | +| Mind2Web browser TYPE | 2.66% | Producing necessary workflow text | +| HelpSteer3 principle | 21.85% | Applying a stated natural-language principle | +| Schema-Guided Dialogue | 8.62% | Decisions conditioned on dialogue and workflow state | +| GoEmotions attribute Score | 8.40% | Judging whether a specified attribute applies | +| BoolQ | 10.09% | Evidence-grounded yes/no decisions | +| MNLI | 16.81% | Entailment, contradiction, and neutral relations | +| Local counterfactual records | 1.34% | Sensitivity to decision-relevant factual changes | + +Source percentages are calculated over the main training schedule and rounded to two decimal places. Their displayed sum may differ slightly from 100% because of rounding. + +The common candidate interface makes these tasks compatible without erasing their differences. Browser examples ground choices in visible elements and goals; principle examples require conditioning on the rule currently provided; entailment and binary questions emphasize relations between claims and evidence. Counterfactual records target changes that should alter a decision, rather than encouraging invariance to every input perturbation. + +Data conversion, candidate construction, parent grouping, and training manifests preserve the relationship between source evidence and target decisions. The main schedule is a heterogeneous task mixture; starter records and main records are tracked separately, including potential overlap. + +## 5. Jet-DCRL: Decision Utility and Probability Quality + +A useful next step is to distinguish three properties that accuracy alone cannot express: whether a decision is correct, whether its probability reflects uncertainty, and whether its consequence is valuable under the application's costs. We call this research extension **Jet-DCRL — Decision-Calibrated Reinforcement Learning**. + +Its candidate-space objective combines proper-scoring components, reference preservation, and decision utility: + +`L_DCRL = mean_d [NLL(q, p_d) + λ_B Brier(q, p_d)]` + +` + λ_ref KL(p_high || p_reference)` + +` + λ_U L_utility(p_low)`. + +The NLL and Brier terms supervise probabilities at both exits. The reference term anchors the deeper distribution to a frozen reference. The utility term targets the earlier exit, where better decisions have particular value for low-cost execution. Candidate utilities are supplied as verified training information; they are not inferred merely from the model's confidence. + +For bounded candidate sets, the exact utility objective is tractable: + +`L_utility = −Σ_a p_low(a) u(a)`. + +A sampled alternative draws actions from the current shallow policy and uses an expected-utility baseline in its policy-gradient estimator. Comparing exact and sampled optimization is valuable here: when the action space is small, sampling should earn its complexity through evidence rather than being assumed necessary. + +The research objective and supporting implementation are separate from the released models: **the published weights in this report have not been trained with Jet-DCRL**. Its purpose is to test whether utility-aware learning can improve economical decisions while preserving probability quality and deeper-path behavior. Independent NLL, Brier, reliability, and risk–coverage measurements will assess the resulting probabilities. + +## 6. Performance: Where the Savings Can Come From + +The framework offers complementary opportunities to reduce decision cost. First, a bounded decision can be obtained from candidate scores without autoregressively generating an entire structured document. Second, an earlier exit executes fewer backbone blocks. Third, selected candidate projections can reduce output-head work where the execution path supports them. These mechanisms address different components of inference and should be measured separately before their gains are combined. + +The cost model includes context processing, executed depth, output projection, and any additional queries sharing the context. Shared-context execution offers a further opportunity to amortize input processing across related decisions. Its design must preserve both attention caches and the backbone's recurrent state, with numerical equivalence established before performance measurements. + +Our performance target is the operating curve between useful decisions and deployment cost: end-to-end latency, sustainable request rate, memory demand, and task success under a declared workload. Local model-forward timings and remote HTTP response times are reported separately; the next deployment comparison will align request boundaries and workloads. + +## 7. Initial Evidence and the Evaluation Agenda + +### 7.1 Decision accuracy + +
+Decision accuracy: APUS-OpenJev 9B 85.00%, APUS-OpenJev 4B and Jev 82.50%, Laya complete-input 68.75%. +
Figure 3. Decision accuracy on the same frozen panel. APUS results use the high compute budget; Laya uses the complete-input typed configuration after truncation was corrected.
+
+ +| System | Decision accuracy | +|---|---:| +| APUS-OpenJev-v1 4B | 82.50% | +| APUS-OpenJev-v1 9B | 85.00% | +| Jev official API | 82.50% | +| Laya | 68.75% | + +These results use a frozen development comparison panel of 80 questions across 79 parent groups, with 16 questions each from Browser, HelpSteer3, BoolQ, MNLI, and Score. APUS results use the high compute budget; the 9B entry uses the selected intermediate version, and Laya uses its typed, complete-input evaluation with truncation corrected. The panel has been used during development and version selection. Its Score subset contains only No labels, and browser questions evaluate static candidate selection rather than completed browsing episodes. Historical latency results distinguish local inference from the official API's complete HTTP response. + +### 7.2 Response latency + +
+P50 and P95 decision latency for APUS-OpenJev 9B, 4B, Laya and Jev, with measurement boundaries labeled. +
Figure 4. Historical decision latency, showing P50 and P95 on separately labeled scales. Local model forward, local pipeline, and complete public HTTP response include different work; these observations do not establish a direct speedup ratio.
+
+ +| Measured path | P50 | P95 | Timing boundary | +|---|---:|---:|---| +| APUS-OpenJev 9B reference | 77.76 ms | 505.16 ms | Local model forward | +| APUS-OpenJev 4B reference | 85.43 ms | 406.30 ms | Local model forward | +| Laya typed, complete input | 8.68 ms | 37.66 ms | Local pipeline | +| Jev official API | 436.96 ms | 4874.61 ms | Complete public HTTP response | + +APUS timings come from the historical RTX PRO 6000 reference implementation. Laya includes tokenization, transfer, model execution, and output processing; its displayed statistics are the medians of three per-run P50 and P95 values. Jev includes network, queueing, and service overhead. The measurements describe complete decisions rather than streaming first-token latency or generated tokens per second. Exact values, source identities, and aggregation are retained in the accompanying `chart-data.json`. + +### 7.3 Evaluation agenda + +The next evaluation phase should connect each architectural claim to an appropriate experiment. Depth learning requires matched initialization and data controls against full-depth-only training. Routing requires frozen policies, per-task risk–coverage curves, and a measured relationship between saved computation and errors. Shared-state execution requires numerical and decision-equivalence checks before latency tests. Browser capability requires replayable environments, action and TYPE evaluation, recovery behavior, and complete task outcomes. Jet-DCRL requires comparisons against continued supervision and exact utility optimization, with probability metrics reported alongside accuracy. + +APUS-OpenJev-v1 establishes a concrete foundation for this program: language-conditioned decisions, shared parameters across execution budgets, and a training objective that connects those budgets. Its central direction is to make computational effort a controllable dimension of useful language-model behavior, evaluated through decisions and the workflows they enable. + +## Acknowledgments + +We thank the Qwen team for the Qwen3.5-4B and Qwen3.5-9B base models, the ms-swift contributors for the training framework, and the creators of the open datasets used in this work. diff --git a/TECHNICAL_REPORT.pdf b/TECHNICAL_REPORT.pdf new file mode 100644 index 0000000000000000000000000000000000000000..c521fff4d42c3e7e5da5134f849cc00d092dd7ce --- /dev/null +++ b/TECHNICAL_REPORT.pdf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6bba2af70bebb6e2dff1c66ed6171332be7f9842adaa4489490bd2065b8a989f +size 709307 diff --git a/assets/accuracy.svg b/assets/accuracy.svg new file mode 100644 index 0000000000000000000000000000000000000000..0e595a342d8e3081542b95e9afac64d88de82b77 --- /dev/null +++ b/assets/accuracy.svg @@ -0,0 +1 @@ +Decision accuracyAPUS 9B 85%, APUS 4B 82.5%, Jev API 82.5%, Laya typed full input 68.75%. Reused development panel, not independent final test.Clear, verifiable decisions.APUS models: full compute budget | Laya: input truncation fixed0%25%50%75%100%APUS-OpenJev 9B85% · 68/80APUS-OpenJev 4B82.5% · 66/80Jev official API82.5% · 66/80Laya typed / full input68.75% · 55/8080 questions / 79 groups · 5 tasks × 16 · Score subset: all NoDevelopment regression only. +2 answers is not proof of general superiority.Laya full-input typed: 55/80 in all three independent-process repeats. Data: chart-data.json \ No newline at end of file diff --git a/assets/compute-depth.svg b/assets/compute-depth.svg new file mode 100644 index 0000000000000000000000000000000000000000..2a8203112ef7dd27cca256f974e9f52fd52b52c3 --- /dev/null +++ b/assets/compute-depth.svg @@ -0,0 +1 @@ +Spend computation where the application needs it.A semantic decision request is compiled once, an effort budget selects either a true early exit or the complete backbone, and the selected candidate is returned as structured data. Calibrated adaptive continuation is a separate research direction.APUS–OPENJEV · ARCHITECTURE STUDYSpend computation where the application needs it.One model family · selectable execution depth · bounded decisions without a text-decoding loopSAME BACKBONE PARAMETERS, TWO EXECUTION PATHSDecision requestContext + instructionRequest-defined candidatesSemantic compilationCandidate-to-token mappingOne verified answer boundaryCompute budgeteffort = low / highApplication-selectedcost–quality balanceLow: execute to depth dShared norm + candidate rowsRead the early decisionUpper blocks are skippedNo execution beyond the exitHigh: lower segmentExecute blocks 1 … dRetain full semantic capacityHigh: remaining blocksExecute blocks d+1 … DNative head → candidate selectionStructured decisionSoftmax over valid candidatesSelect a candidate / decision valueHost code assembles the responseEach request follows one selected path; the diagram shows alternative execution budgets.RESEARCH EXTENSION · CALIBRATED PER-REQUEST DEPTHEarly distribution + task featuresMeasure uncertainty on held-out tasks.Calibrated acceptance policyExit, deepen, observe, or abstain.Continue residualIntegrate tested depth primitive.Solid flow: release execution paths Dashed region: proposed serving integration Text generation is a separate path. \ No newline at end of file diff --git a/assets/dataset-subsets.svg b/assets/dataset-subsets.svg new file mode 100644 index 0000000000000000000000000000000000000000..f444e8c69a28c6250e5ec44345da0d1bd1fabd20 --- /dev/null +++ b/assets/dataset-subsets.svg @@ -0,0 +1 @@ +Frozen80 evaluation subset compositionFive subsets each contain 16 questions; Browser, HelpSteer3, BoolQ and MNLI each have 16 parent groups, while Score has 15. Total 80 questions and 79 groups.Frozen80: five decision capabilities80 questions / 79 parent groups / the same frozen IDs for every modelBrowser16 questions · 20%HelpSteer316 questions · 20%BoolQ16 questions · 20%MNLI16 questions · 20%Score16 questions · 20%SUBSET & SOURCETARGET CAPABILITYQUESTIONS / GROUPSBrowserOSU NLP / Mind2WebSelect the next browser actionfrom page context and a goal.16 / 16HelpSteer3NVIDIA / principle-based judgmentsJudge content against a suppliednatural-language principle.16 / 16BoolQBoolean QuestionsAnswer yes/no questionsusing supporting context.16 / 16MNLIMulti-Genre NLIIdentify entailment, contradictionor a neutral relationship.16 / 16ScoreGoogle / GoEmotionsJudge whether a specifiedemotion attribute is present.16 / 15Scope: static browser actions, principle judgments, contextual reasoning and attribute decisions.Score has 16 No labels and no Yes labels. This is a development regression panel, not a final test.Source: subset-chart-data.json · Equal task sizes do not imply equal task difficulty. diff --git a/assets/latency.svg b/assets/latency.svg new file mode 100644 index 0000000000000000000000000000000000000000..c40ebba253044c716848679c26ec440e2956502f --- /dev/null +++ b/assets/latency.svg @@ -0,0 +1 @@ +Response latency bars with distinct timing boundariesTwo horizontal bar panels. P50 scale 0 to 500 milliseconds; P95 scale 0 to 5000 milliseconds. APUS4B 85.43 and 406.30ms; APUS9B 77.76 and 505.16ms; Jev HTTP 436.96 and 4874.61ms; Laya full input 8.68 and 37.66ms. Different timing boundaries cannot support speedup ratios.Response latency — see the numbers and their boundaries.Same fixed 80-question panel · horizontal bars start at zero · lower measured time = shorter barP50 · typical requestLinear scale: 0–500 ms0100200300400500msAPUS-OpenJev 4BLocal model forward / historical85.43APUS-OpenJev 9BLocal model forward / historical77.76Jev official APIRemote HTTP / network + queue + service436.96Laya typed / full inputLocal system_one / 3-run median8.68P95 · slower tailLinear scale: 0–5,000 ms01,0002,0003,0004,0005,000msAPUS-OpenJev 4BLocal model forward / historical406.30APUS-OpenJev 9BLocal model forward / historical505.16Jev official APIRemote HTTP / network + queue + service4,874.61Laya typed / full inputLocal system_one / 3-run median37.66Different timing scopes and runs: do not divide these bars to claim a speedup factor.APUS: historical forward timings, not a new merged-package test. Laya: median of 3 per-run P50/P95 values.P50 and P95 panels use different scales. Not text TTFT or tokens/s. Exact data, scope and source hashes: chart-data.json \ No newline at end of file diff --git a/assets/learning-framework.svg b/assets/learning-framework.svg new file mode 100644 index 0000000000000000000000000000000000000000..5e828aeef04e2efe6494d92c5ff81eb0eaf429af --- /dev/null +++ b/assets/learning-framework.svg @@ -0,0 +1 @@ +Learning decisions across compute budgets.Training reuses one native backbone and one normalization and output vocabulary head at two depths, with supervised decision losses and stop-gradient deep-to-shallow distillation; Jet-DCRL is a separate research training stage.APUS–OPENJEV · ARCHITECTURE STUDYLearning decisions across compute budgets.Candidate semantics enter as language; depth becomes a learned decision interface.SHARED SEMANTIC BACKBONE · ONE FORWARD TRAVERSALSemantic task inputContext + questionCandidate descriptionsShared lower segmentNative blocks 1 … dRetain the residual state h(d)Remaining upper segmentNative blocks d+1 … DContinue from h(d), without replayLow-budget decision distributionShared final norm + vocabulary headProject candidate rows at answer positionp(low) over the request’s candidatesHigh-budget decision distributionSame norm and output-head parametersSame candidate semantics and target spacep(high) also provides a training targetsg(·)LANGUAGE RETENTIONOpen-ended TYPE examplesuse full-path token CE.Interleaved training batchespreserve a text output path.Labels supervise both exits; gold answers are not prompt tokens.Joint decision learningL = 0.5 CE(target, p(low)) + 0.5 CE(target, p(high)) + 0.1 KL(stopgrad[p(high)] || p(low))MULTI-OBJECTIVEDecision supervision+ text retentionThe deep distribution teaches the early exit; both exits remain anchored to the supervised target.TRAINING PROGRAM · RELEASE LINEAGE AND NEXT STAGEInitialization4B: decision SFT warm-start9B: base initializationSeparate model lineagesJoint supervised curriculumDual-exit CE + cross-depth KLInterleaved full-path text retentionImplemented release trainingJet-DCRL · research stageDecision-Calibrated Reinforcement LearningDecision utility + probability quality + reference anchoringProposed follow-on training; evaluate calibrated budget policiesCE and cross-depth distillation are one joint optimization objective, not separate completed stages.Solid arrows: implemented flow Dashed teacher arrow: stop-gradient Dashed stage: research program \ No newline at end of file diff --git a/assets/task-accuracy.svg b/assets/task-accuracy.svg new file mode 100644 index 0000000000000000000000000000000000000000..a3c74929e4010daa42cf4040f03c3bbf2a991245 --- /dev/null +++ b/assets/task-accuracy.svg @@ -0,0 +1 @@ +Frozen80 per-task model accuracyPer-task exact-match accuracy of APUS-OpenJev 4B, APUS-OpenJev 9B, Jev official API and Laya typed full-input. Overall scores are 66, 68, 66 and 55 out of 80.Decision accuracy by capabilityMatched questions · APUS-OpenJev full compute budget · Laya full input, zero truncationSUBSETAPUS-OpenJev4BAPUS-OpenJev9BJevofficial APILayatyped / full inputBrowser16 questions / 16 groups14 / 1687.50%13 / 1681.25%11 / 1668.75%4 / 1625.00%HelpSteer316 questions / 16 groups12 / 1675.00%13 / 1681.25%12 / 1675.00%11 / 1668.75%BoolQ16 questions / 16 groups16 / 16100.00%16 / 16100.00%16 / 16100.00%15 / 1693.75%MNLI16 questions / 16 groups11 / 1668.75%13 / 1681.25%13 / 1681.25%11 / 1668.75%Score16 questions / 15 groups13 / 1681.25%13 / 1681.25%14 / 1687.50%14 / 1687.50%Overall80 questions / 79 groups66 / 8082.50%68 / 8085.00%66 / 8082.50%55 / 8068.75%Each subset has 16 questions: one answer changes accuracy by 6.25 percentage points.Score is all No (always-No baseline: 16/16). Browser is static action selection, not live task success.Laya: identical decisions across 3 process repeats. Reused development panel; not a superiority claim.Cell shading and bars share a 0–100% scale. Source: subset-chart-data.json diff --git a/bundle-manifest.json b/bundle-manifest.json new file mode 100644 index 0000000000000000000000000000000000000000..78dc88e1ffc78f654c737ecfcb9159c1029d2650 --- /dev/null +++ b/bundle-manifest.json @@ -0,0 +1,639 @@ +{ + "format": "apus-openjev.organization-mirror.v1", + "repo_id": "apus-ailab/APUS-OpenJev-v1", + "private": false, + "source_repo": "gump2049/APUS-OpenJev-v1", + "source_revision": "e7e3cc0b9c82b91380ec6595120b7ce8abd32fd7", + "source_bundle_manifest_sha256": "73832c9880cd69b256fbae6914a0f04e62d59f5529d89d6d906863072ea067ab", + "new_training_or_gpu_evaluation": false, + "transfer": "Official CommitOperationCopy with src_repo_id; server-side LFS copy", + "editorial_overrides": [ + "4B-5949/README.md", + "4B-5949/release-manifest.json", + "9B-3000/README.md", + "9B-3000/release-manifest.json", + "9B-5949/README.md", + "9B-5949/release-manifest.json", + "ARCHITECTURE.md", + "README.md", + "README.zh-CN.md" + ], + "files": [ + { + "path": "4B-5949/LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "4B-5949/README.md", + "bytes": 1924, + "sha256": "4071763e7fbd1a16910c7bd8d7b6751e09b999f98bcdbe1d2f118f6b10f0c40f" + }, + { + "path": "4B-5949/RUNTIME.md", + "bytes": 5556, + "sha256": "d9397e883c8b8709297f394c7c65eb0ee6b17c0994902b6f6d518d924fd2c854" + }, + { + "path": "4B-5949/chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "4B-5949/config.json", + "bytes": 2829, + "sha256": "4b53bdb886bb68d228841409c2db2f5d7eb670c28ae4376849f58d3bd014968d" + }, + { + "path": "4B-5949/decision-release.json", + "bytes": 1486, + "sha256": "51f8cd280dae539bc8cf1055b6a6ae64b8dfb13064ec3d155cf40ebe9079c690" + }, + { + "path": "4B-5949/depth_config.json", + "bytes": 320, + "sha256": "50a4f97bbb3589222d81285ff94b33efd0f048d159166f6b93f2700e633be3e0" + }, + { + "path": "4B-5949/evaluation/runtime-smoke.json", + "bytes": 4858, + "sha256": "60a80902675971ff6a054377852b936eec90f49f466eb7b8b077eee7b979f230" + }, + { + "path": "4B-5949/evaluation/weight-arithmetic.json", + "bytes": 25495, + "sha256": "567c5c0f85e9b1db314c58c896542eaf0753d0723fa94e61f134660e813ea18d" + }, + { + "path": "4B-5949/examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "4B-5949/generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "4B-5949/merge-provenance.json", + "bytes": 682, + "sha256": "3dcce1f04722cfd98a30c060b96eb1fc6b8b1d321904486acf1c1b2693c66d3a" + }, + { + "path": "4B-5949/merged-evaluation.json", + "bytes": 3399, + "sha256": "b653bb0672e530a01ac58ab1b4597f7a9c81ddc157db2b0fdc8cf64ae73d85ed" + }, + { + "path": "4B-5949/model-00001-of-00003.safetensors", + "bytes": 3991298872, + "sha256": "c04bef62040612a3376c144014c194687cdc19b18c3a807ee1136b4a713cbb0a" + }, + { + "path": "4B-5949/model-00002-of-00003.safetensors", + "bytes": 3979833152, + "sha256": "7891d5b76b686893680e9cc074c2e17a788ff0cb03f64cc2ae4150bd804299a2" + }, + { + "path": "4B-5949/model-00003-of-00003.safetensors", + "bytes": 1107487880, + "sha256": "a88eecd668f83cad773bd67c9c7e6e466c1746d489c55d906b48398a6679db65" + }, + { + "path": "4B-5949/model.safetensors.index.json", + "bytes": 66236, + "sha256": "cd67b86e2cd9d329167224ac0026be69b39e078ad3056fc59339ba0a19fe4845" + }, + { + "path": "4B-5949/openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "4B-5949/openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "4B-5949/openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "4B-5949/openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "4B-5949/openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "4B-5949/processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "4B-5949/provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "4B-5949/provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "4B-5949/provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "4B-5949/provenance/identity.json", + "bytes": 682, + "sha256": "3dcce1f04722cfd98a30c060b96eb1fc6b8b1d321904486acf1c1b2693c66d3a" + }, + { + "path": "4B-5949/release-manifest.json", + "bytes": 6364, + "sha256": "86abf236e99c645e98790eb19c020af4b2790fb199929e24a535dec2cf8eba5b" + }, + { + "path": "4B-5949/requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "4B-5949/runtime-validation.json", + "bytes": 334, + "sha256": "96d38fd523772afe5cc96e645fa2bc14d20e58a1920da3822e6c42b6cfeaf3e4" + }, + { + "path": "4B-5949/source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "4B-5949/tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "4B-5949/tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "4B-5949/training.md", + "bytes": 2692, + "sha256": "30544dc7ca3119ff10e88507651a3275b62f61b7e888f3332eb8fab632caae6b" + }, + { + "path": "9B-3000/LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "9B-3000/README.md", + "bytes": 1924, + "sha256": "8e6939c48caf1cd839c9c52c41ed7c766d79abcab255a697a8e859ea19d22f0d" + }, + { + "path": "9B-3000/RUNTIME.md", + "bytes": 5556, + "sha256": "d9397e883c8b8709297f394c7c65eb0ee6b17c0994902b6f6d518d924fd2c854" + }, + { + "path": "9B-3000/chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "9B-3000/config.json", + "bytes": 2832, + "sha256": "b1d02fb6b40aeba43105ca558b2532f31d062e0bd95bce564d9996c0a34b4211" + }, + { + "path": "9B-3000/decision-release.json", + "bytes": 1484, + "sha256": "01b56794252336f6a3385a06cd63688216039e125cb572ae6c0ca0675228a593" + }, + { + "path": "9B-3000/depth_config.json", + "bytes": 320, + "sha256": "f86116c627cd77eee275a9d1632b6783b085171ad85ade66699d266c432ab943" + }, + { + "path": "9B-3000/evaluation/runtime-smoke.json", + "bytes": 4905, + "sha256": "70d686828090965063966dddac6d8454cf981c6614a406c1647617986e534c5d" + }, + { + "path": "9B-3000/evaluation/weight-arithmetic.json", + "bytes": 25495, + "sha256": "efc2c717c9c8d62fec62ba9236db8f29166111961ade2f99d038b8c37aac54d4" + }, + { + "path": "9B-3000/examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "9B-3000/generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "9B-3000/merge-provenance.json", + "bytes": 682, + "sha256": "3f6af915e561ce7fe67ec6ba64bf1e419923e23c609d71db4cf63aded0b202b3" + }, + { + "path": "9B-3000/merged-evaluation.json", + "bytes": 3395, + "sha256": "e0683fb077b2af2ca4193950a211c5e0a31560fb003b44c4c43e688248fa3587" + }, + { + "path": "9B-3000/model-00001-of-00006.safetensors", + "bytes": 2034237568, + "sha256": "dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344" + }, + { + "path": "9B-3000/model-00002-of-00006.safetensors", + "bytes": 3999615808, + "sha256": "3fe0ee3d29088f7f84ac7ba8c9d7562b91e6a31841c5b585c71779e618519e99" + }, + { + "path": "9B-3000/model-00003-of-00006.safetensors", + "bytes": 3997274128, + "sha256": "7bfce51b034a6de02c513b032c97007532676c5f915d020aa9a66f399f437117" + }, + { + "path": "9B-3000/model-00004-of-00006.safetensors", + "bytes": 3997290904, + "sha256": "f36d2b3ffc4c44dab06277ebbd722f73faca80aefc9aa3ce1119575f4a4773ae" + }, + { + "path": "9B-3000/model-00005-of-00006.safetensors", + "bytes": 3991239264, + "sha256": "b06602342e98eb882703857f3c8e8058894c03d4f2d62ddaab1e6dd4913c4e9b" + }, + { + "path": "9B-3000/model-00006-of-00006.safetensors", + "bytes": 800062816, + "sha256": "db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013" + }, + { + "path": "9B-3000/model.safetensors.index.json", + "bytes": 69253, + "sha256": "3d2f0ab780828a41449b9f32fee63e6ee0d27bf98fab0e52aa982f81e6c47cec" + }, + { + "path": "9B-3000/openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "9B-3000/openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "9B-3000/openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "9B-3000/openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "9B-3000/openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "9B-3000/processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "9B-3000/provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "9B-3000/provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "9B-3000/provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "9B-3000/provenance/identity.json", + "bytes": 682, + "sha256": "3f6af915e561ce7fe67ec6ba64bf1e419923e23c609d71db4cf63aded0b202b3" + }, + { + "path": "9B-3000/release-manifest.json", + "bytes": 6882, + "sha256": "8da62c69f2d6e40efa17cab0eb18bc178e90b211cdfe0497aff048da3d0877fd" + }, + { + "path": "9B-3000/requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "9B-3000/runtime-validation.json", + "bytes": 334, + "sha256": "c291e6528feb11e87b878e6ea89676b2ef9fcb1cda7403efece382d891bb2755" + }, + { + "path": "9B-3000/source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "9B-3000/tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "9B-3000/tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "9B-3000/training.md", + "bytes": 2821, + "sha256": "0918aefaf0bf925ebf9aa7cbb9bbaa65ab5102de889ec0793b945f1936d84a0e" + }, + { + "path": "9B-5949/LICENSE", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "9B-5949/README.md", + "bytes": 1941, + "sha256": "fa8d623c896606d578e3e8dbcad687ab9ef490811dfd9269c04b0570c008d546" + }, + { + "path": "9B-5949/RUNTIME.md", + "bytes": 5557, + "sha256": "5e353f905571c015d655d9f2e46cbcfab31028ad45c5261aa7bba4dcf59d7275" + }, + { + "path": "9B-5949/chat_template.jinja", + "bytes": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + { + "path": "9B-5949/config.json", + "bytes": 2832, + "sha256": "b1d02fb6b40aeba43105ca558b2532f31d062e0bd95bce564d9996c0a34b4211" + }, + { + "path": "9B-5949/depth_config.json", + "bytes": 320, + "sha256": "7729f1ed0408aeeff51718f765fd017b9e2e61fe1fd329b78abeb9cbb42a6129" + }, + { + "path": "9B-5949/evaluation/runtime-smoke.json", + "bytes": 4923, + "sha256": "1a9e55991d3a91dc4c704b080f6e251f3be244bfddb56246a0bdbcb00eb278b1" + }, + { + "path": "9B-5949/evaluation/weight-arithmetic.json", + "bytes": 25500, + "sha256": "3121372fff18946ad31f3d750f52d55c9f53516a0fb06aa525f4a9c3a94a1757" + }, + { + "path": "9B-5949/examples.py", + "bytes": 2332, + "sha256": "982ced288cae6ccf70ba70174f3c485398a8e712597b5b85423ba9b84318bf39" + }, + { + "path": "9B-5949/generation_config.json", + "bytes": 116, + "sha256": "e4b598e9544d7567b3ae288efd8417e5b1646c957609b614c639163139336d12" + }, + { + "path": "9B-5949/merge-provenance.json", + "bytes": 682, + "sha256": "9e196f258b24eff17da41e08bc3ebf5914e3b2edad5196e9146f4d9286d6a0d9" + }, + { + "path": "9B-5949/merged-evaluation.json", + "bytes": 3446, + "sha256": "6bc22a9795815eb46f07f4c83e2ac20a861afbd7b1099e82b5f810ed24b4db9b" + }, + { + "path": "9B-5949/model-00001-of-00006.safetensors", + "bytes": 2034237568, + "sha256": "dd63614f1dc80dce2d83be3f3e69af1f8a0ebc8add9b8e0abf3910f563e44344" + }, + { + "path": "9B-5949/model-00002-of-00006.safetensors", + "bytes": 3999615808, + "sha256": "7dfbf5a64c3b383dfa996f7c438db87b6545453ad3092f1798005c0de4ad8299" + }, + { + "path": "9B-5949/model-00003-of-00006.safetensors", + "bytes": 3997274128, + "sha256": "f1f2e7d9e1309ff2207a6004f4f749f6e7fc67ac303423e1a9c8eaaf381916d6" + }, + { + "path": "9B-5949/model-00004-of-00006.safetensors", + "bytes": 3997290904, + "sha256": "db2c6e953a2899b1410b0bba753c1f29b618d5c5303d51c57300949f27f2a01e" + }, + { + "path": "9B-5949/model-00005-of-00006.safetensors", + "bytes": 3991239264, + "sha256": "cde34ae4e7a343a8ad96deee03516d0bdf3197bb6132727f47e8ada4a862b284" + }, + { + "path": "9B-5949/model-00006-of-00006.safetensors", + "bytes": 800062816, + "sha256": "db688e61fd7575e8537120bc7aaeff043d31d6ca2bafa14df1135c969a081013" + }, + { + "path": "9B-5949/model.safetensors.index.json", + "bytes": 69253, + "sha256": "3d2f0ab780828a41449b9f32fee63e6ee0d27bf98fab0e52aa982f81e6c47cec" + }, + { + "path": "9B-5949/openjet_runtime/__init__.py", + "bytes": 52, + "sha256": "df6eb864cf6d0c2f512fb00f17eeaa11fd790afca33a0f7b2f609aeb1f2b3944" + }, + { + "path": "9B-5949/openjet_runtime/candidate_projection.py", + "bytes": 4438, + "sha256": "84dc4746b5fb06ac6a9024dde3ba8414d901acf2a62d010b0d66f26acfaf74a6" + }, + { + "path": "9B-5949/openjet_runtime/contracts.py", + "bytes": 3847, + "sha256": "d8e8e5270ecd6dab917d886dda2d684c24696b811faa968bb3c399ceef5e356a" + }, + { + "path": "9B-5949/openjet_runtime/early_exit.py", + "bytes": 7862, + "sha256": "89b7751a927a6d1348a454fb8ee986395423ad5686d855fb3ccab255c95d2566" + }, + { + "path": "9B-5949/openjet_runtime/runtime.py", + "bytes": 8081, + "sha256": "6e7b0b131cb14ab0d25cc8fd6c7fc41738b799cfe6de1ccdbae2d09d7c64313c" + }, + { + "path": "9B-5949/processor_config.json", + "bytes": 1220, + "sha256": "bfbc24af59a3e73a9cd0653b8d4ae758dfaec4e3e6c15dfb1c6c8ee8d5683c85" + }, + { + "path": "9B-5949/provenance/BASE-LICENSE.txt", + "bytes": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + { + "path": "9B-5949/provenance/SOURCE-PROJECT-LICENSE.txt", + "bytes": 11357, + "sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4" + }, + { + "path": "9B-5949/provenance/code-license-provenance.json", + "bytes": 397, + "sha256": "33a6e44402d44b5febe1451ec4978523858fd4e91f5f101e61ad9b39e9b21f2f" + }, + { + "path": "9B-5949/provenance/identity.json", + "bytes": 682, + "sha256": "9e196f258b24eff17da41e08bc3ebf5914e3b2edad5196e9146f4d9286d6a0d9" + }, + { + "path": "9B-5949/release-manifest.json", + "bytes": 6887, + "sha256": "c6bcfcc50262ec732bb27d5fc5cb169f7a6dfdeac7d68017b5238e91be9765e1" + }, + { + "path": "9B-5949/requirements.txt", + "bytes": 139, + "sha256": "a5ceffcb009f4fe48882b20ef19d161ce53678abb5234175f45a8c82c3c868c6" + }, + { + "path": "9B-5949/reviewed-variant.json", + "bytes": 1254, + "sha256": "7c134fe8297dd19227aea63ba8cc5cade5ab6aa380569dcf020ab32b912b885a" + }, + { + "path": "9B-5949/runtime-validation.json", + "bytes": 334, + "sha256": "83b021ff809c981760e56667355e63c29806b909b96eebb1f876f4f63c06b150" + }, + { + "path": "9B-5949/source-provenance.json", + "bytes": 773, + "sha256": "a623caf50f658bee5c1ad5a084e486a945e288fa93cdb798f89a58d0c5e050c4" + }, + { + "path": "9B-5949/tokenizer.json", + "bytes": 19989325, + "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" + }, + { + "path": "9B-5949/tokenizer_config.json", + "bytes": 1165, + "sha256": "792fa3f0cb88b111e54ef3134c873531008c4df471d108da17903426e308aa7b" + }, + { + "path": "9B-5949/training.md", + "bytes": 2697, + "sha256": "c03d263c6f8b8abeac60cade36b5cde0667b75b5e26ddfdbfb8ab49361e2cdaf" + }, + { + "path": "ARCHITECTURE.md", + "bytes": 5311, + "sha256": "263a37bc7be91ab4321a6728ccb8fab17e47bd7f0821cfaaa691047725b8db14" + }, + { + "path": "LICENSE", + "bytes": 1109, + "sha256": "a0a6df200500edc4f07c7238061368631a0ac90c22fe2af6c658b2c3629611c9" + }, + { + "path": "LICENSE_NOTICES.md", + "bytes": 1169, + "sha256": "8eb4d63c9506cd955a889048f2fa331a14fc77df23818c9c0e492b9e79a543ce" + }, + { + "path": "README.md", + "bytes": 10985, + "sha256": "b8e236d48e687c680faa1c9c4856e1206887e9c3287349e12bc88c67d333f35f" + }, + { + "path": "README.zh-CN.md", + "bytes": 9906, + "sha256": "2c3c416100dca3c799d964fd1ca88eddbe6ad7f861cb7166e8368120fd7ae3d2" + }, + { + "path": "TECHNICAL_REPORT.md", + "bytes": 19474, + "sha256": "3054ed2a1bf18d5aacd1691bc9aac0ba1a7fe8b20eefb0a8c7cd1c301f444fc2" + }, + { + "path": "TECHNICAL_REPORT.pdf", + "bytes": 709307, + "sha256": "6bba2af70bebb6e2dff1c66ed6171332be7f9842adaa4489490bd2065b8a989f" + }, + { + "path": "assets/accuracy.svg", + "bytes": 2913, + "sha256": "63499929bbee87b29186436c62c06c801f6356c89c2ca18f9640f632959f9f16" + }, + { + "path": "assets/compute-depth.svg", + "bytes": 8928, + "sha256": "ebf81e08c914bb0730e06cf40c24f4bcd9e1fa8990d72aa9527049c3b07c8b6b" + }, + { + "path": "assets/dataset-subsets.svg", + "bytes": 6425, + "sha256": "b3198ec01536b6efff1c8f89402d3b45112c21471d613ed69f9a5c1a5fb94762" + }, + { + "path": "assets/latency.svg", + "bytes": 6956, + "sha256": "225519f2085272a38f26021baab64df5f9550f0764257c6724d83db48ca34ee5" + }, + { + "path": "assets/learning-framework.svg", + "bytes": 10758, + "sha256": "b64f4032b655cc0af1ed83e51cac58eb2d4069058b5ef6d948c024767636e38c" + }, + { + "path": "assets/task-accuracy.svg", + "bytes": 15240, + "sha256": "0390e44ff2cef30c117d74b95b55384141592ed545e8f3cc48e87cd630c3165d" + }, + { + "path": "chart-data.json", + "bytes": 4501, + "sha256": "ce8ef4042c83f39b43b334b7918b8c820e4188391fe638fa1f74064ef621db99" + }, + { + "path": "subset-chart-data.json", + "bytes": 58611, + "sha256": "ea3f366d22f7e761901343dd494eb809e31f71813dd4ae0194a0bf7aabe888fe" + } + ], + "manifest_self_excluded": true +} diff --git a/chart-data.json b/chart-data.json new file mode 100644 index 0000000000000000000000000000000000000000..b877a0a26e4f61583ced14b842f43ca6fa3df576 --- /dev/null +++ b/chart-data.json @@ -0,0 +1,130 @@ +{ + "title": "APUS-OpenJev-v1", + "created_date": "2026-09-21", + "panel": { + "records": 80, + "parent_groups": 79, + "task_counts": { + "Browser": 16, + "HelpSteer3": 16, + "BoolQ": 16, + "MNLI": 16, + "Score": 16 + }, + "score_labels": "16 No, 0 Yes", + "split": "validation/development regression; not sealed final", + "laya_export_sha256": "26aa47e8f56c18a403eb4c8379d5cccb9cea5af97c191ab26ec36b8fb17ae923" + }, + "accuracy": [ + { + "model": "APUS-OpenJev 9B", + "correct": 68, + "total": 80, + "percentage": 85.0, + "source": "merged9b", + "effort": "high", + "variant_directory": "9B-3000" + }, + { + "model": "APUS-OpenJev 4B", + "correct": 66, + "total": 80, + "percentage": 82.5, + "source": "merged4b", + "effort": "high", + "variant_directory": "4B-5949" + }, + { + "model": "Jev official API", + "correct": 66, + "total": 80, + "percentage": 82.5, + "source": "native9b_latency" + }, + { + "model": "Laya typed / full input", + "correct": 55, + "total": 80, + "percentage": 68.75, + "source": "laya_repeats", + "config": "typed-fullinput", + "repetitions": 3, + "identical_accuracy": true, + "truncations": 0 + } + ], + "latency": { + "native_forward_historical": [ + { + "model": "APUS-OpenJev 9B", + "p50_ms": 77.76357303373516, + "p95_ms": 505.1563191227615, + "source": "native9b_latency" + }, + { + "model": "APUS-OpenJev 4B", + "p50_ms": 85.43, + "p95_ms": 406.3, + "source": "native4b_latency" + } + ], + "laya_local_system_one": { + "model": "Laya typed / full input", + "p50_ms": 8.684632601216435, + "p95_ms": 37.65514073893428, + "p50_range_ms": [ + 8.429150097072124, + 8.730278117582202 + ], + "p95_range_ms": [ + 37.441988941282034, + 37.88864379748702 + ], + "aggregation": "median of three per-run P50 and P95 values; not pooled or fastest-run selection", + "scope": "native system_one includes tokenization, transfer, forward and structured response; excludes HTTP and loading", + "source": "laya_repeats" + }, + "official_http": { + "model": "Jev official API", + "p50_ms": 436.9597500190139, + "p95_ms": 4874.612333020195, + "scope": "remote full HTTP response including network/queue/service; not client-observable TTFT", + "source": "native9b_latency" + }, + "cross_scope_speedup_claim": false, + "merged_package_latency_retest": false + }, + "sources": { + "laya_repeats": { + "project_relative_path": "docs/feat-jev-combined/evidence/laya-repeated-20260921/summary.json", + "sha256": "16a2d3a265909e40f65cdb9331417a6d46a1f2140759f94d2b183728d64825e6" + }, + "native9b_latency": { + "project_relative_path": "docs/feat-jev-combined/evidence/9b-checkpoints80/3000/report.json", + "sha256": "f3b93d04ab8e99d3cf135d410b559e1c96e980fba43c59b71428fe8c815b9303" + }, + "native4b_latency": { + "project_relative_path": "docs/feat-jev-combined/official-typesafe-vs-final5949-80-20260920.md", + "sha256": "5a7568a998d4be13555374399aa91e9b458d782336b14ab33603a2ebcabe99fc" + }, + "merged4b": { + "project_relative_path": "docs/feat-jev-combined/openjet-release-20260921/merged-release/evidence/4B/comparison.json", + "sha256": "97cc94ddebb4f321d3d4ed29c89a0d8b1b3d6570e48bff427c635ba82f67862e" + }, + "merged9b": { + "project_relative_path": "docs/feat-jev-combined/openjet-release-20260921/merged-release/evidence/9B/comparison.json", + "sha256": "d2ea8a7af10f09fdfde166384ffb18fb5a19ebf1a1281c58cb72806a45878c41" + }, + "panel_description": { + "project_relative_path": "docs/feat-jev-combined/openjet-release-20260921/eval-dataset/SPLIT_AND_EVALUATION.md", + "sha256": "b14b8af51f0527443b369f6995d46b786207dbf663ef2ce5afc9cc0a02d31b8b" + } + }, + "limitations": [ + "80 development questions are reused and have influenced development/version selection.", + "Two more correct answers than Jev does not establish significant superiority or noninferiority.", + "Score is all No; Browser is static candidate selection, not website completion.", + "Laya full-input formatting was changed without retraining; three repeats do not increase independent sample size.", + "Historical native forward, local system_one and remote HTTP are different timing boundaries." + ] +} diff --git a/subset-chart-data.json b/subset-chart-data.json new file mode 100644 index 0000000000000000000000000000000000000000..3b07f91539f4c61faadb42f48b933874ffc51eba --- /dev/null +++ b/subset-chart-data.json @@ -0,0 +1,1679 @@ +{ + "title": "Frozen80 subset composition and per-task comparison", + "created_date": "2026-09-21", + "panel": { + "records": 80, + "parent_groups": 79, + "split": "development regression, not sealed final", + "score_labels": "16 No / 0 Yes" + }, + "models": [ + "APUS-OpenJev 4B", + "APUS-OpenJev 9B", + "Jev official API", + "Laya typed / full input" + ], + "model_totals": { + "APUS-OpenJev 4B": { + "correct": 66, + "total": 80, + "percentage": 82.5 + }, + "APUS-OpenJev 9B": { + "correct": 68, + "total": 80, + "percentage": 85.0 + }, + "Jev official API": { + "correct": 66, + "total": 80, + "percentage": 82.5 + }, + "Laya typed / full input": { + "correct": 55, + "total": 80, + "percentage": 68.75 + } + }, + "subsets": [ + { + "dataset": "browser.dev.records.jsonl", + "name": "Browser", + "source": "OSU NLP / Mind2Web", + "target_capability": "Select the next browser action from page context and a goal.", + "records": 16, + "groups": 16, + "percentage_of_panel": 20.0, + "scores": { + "APUS-OpenJev 4B": { + "correct": 14, + "total": 16, + "percentage": 87.5 + }, + "APUS-OpenJev 9B": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "Jev official API": { + "correct": 11, + "total": 16, + "percentage": 68.75 + }, + "Laya typed / full input": { + "correct": 4, + "total": 16, + "percentage": 25.0 + } + } + }, + { + "dataset": "hs3.dev.records.jsonl", + "name": "HelpSteer3", + "source": "NVIDIA / principle-based judgments", + "target_capability": "Judge content against a supplied natural-language principle.", + "records": 16, + "groups": 16, + "percentage_of_panel": 20.0, + "scores": { + "APUS-OpenJev 4B": { + "correct": 12, + "total": 16, + "percentage": 75.0 + }, + "APUS-OpenJev 9B": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "Jev official API": { + "correct": 12, + "total": 16, + "percentage": 75.0 + }, + "Laya typed / full input": { + "correct": 11, + "total": 16, + "percentage": 68.75 + } + } + }, + { + "dataset": "boolq.dev.records.jsonl", + "name": "BoolQ", + "source": "Boolean Questions", + "target_capability": "Answer yes/no questions using supporting context.", + "records": 16, + "groups": 16, + "percentage_of_panel": 20.0, + "scores": { + "APUS-OpenJev 4B": { + "correct": 16, + "total": 16, + "percentage": 100.0 + }, + "APUS-OpenJev 9B": { + "correct": 16, + "total": 16, + "percentage": 100.0 + }, + "Jev official API": { + "correct": 16, + "total": 16, + "percentage": 100.0 + }, + "Laya typed / full input": { + "correct": 15, + "total": 16, + "percentage": 93.75 + } + } + }, + { + "dataset": "mnli.dev.records.jsonl", + "name": "MNLI", + "source": "Multi-Genre NLI", + "target_capability": "Identify entailment, contradiction or a neutral relationship.", + "records": 16, + "groups": 16, + "percentage_of_panel": 20.0, + "scores": { + "APUS-OpenJev 4B": { + "correct": 11, + "total": 16, + "percentage": 68.75 + }, + "APUS-OpenJev 9B": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "Jev official API": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "Laya typed / full input": { + "correct": 11, + "total": 16, + "percentage": 68.75 + } + } + }, + { + "dataset": "score.dev.records.jsonl", + "name": "Score", + "source": "Google / GoEmotions", + "target_capability": "Judge whether a specified emotion attribute is present.", + "records": 16, + "groups": 15, + "percentage_of_panel": 20.0, + "scores": { + "APUS-OpenJev 4B": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "APUS-OpenJev 9B": { + "correct": 13, + "total": 16, + "percentage": 81.25 + }, + "Jev official API": { + "correct": 14, + "total": 16, + "percentage": 87.5 + }, + "Laya typed / full input": { + "correct": 14, + "total": 16, + "percentage": 87.5 + } + } + } + ], + "verification": { + "all_model_exact_ID_sets_match": true, + "all_gold_labels_match": true, + "laya_repeats": 3, + "laya_identical_predictions": true, + "laya_truncated_records": 0, + "laya_raw_choices_reparsed": 240, + "one_answer_percentage_points": 6.25 + }, + "sources": { + "APUS-OpenJev 4B": { + "project_relative_path": "docs/feat-jev-combined/openjet-release-20260921/merged-release/evidence/4B/after.jsonl", + "sha256": "ac22ede19c6ade99388038292644df50e9891dff162acdb10029bb3bd343537e" + }, + "APUS-OpenJev 9B": { + "project_relative_path": "docs/feat-jev-combined/openjet-release-20260921/merged-release/evidence/9B/after.jsonl", + "sha256": "3f01a868d4fba91a8a1749df37a87f10f0874823eec5a978291c93231773c365" + }, + "Jev official API": { + "project_relative_path": "docs/feat-jev-combined/evidence/9b-checkpoints80/3000/predictions.jsonl", + "sha256": "0997f100fd2da6bed882a70c1e5fe7f1d47c8db9c8d79c76bddbf641ce00344d" + }, + "Laya repeat 1": { + "project_relative_path": "docs/feat-jev-combined/evidence/laya-repeated-20260921/round1-typed-fullinput/predictions.jsonl", + "sha256": "abdddedaeb337dbde5f5ffba8205ab14bc3a6d831ddce8a7feca4a97bb9b5b12" + }, + "Laya repeat 2": { + "project_relative_path": "docs/feat-jev-combined/evidence/laya-repeated-20260921/round2-typed-fullinput/predictions.jsonl", + "sha256": "51556848bf798a0fb6f891be400f60c6b8da3505445795ce31d8f60dd2a56a18" + }, + "Laya repeat 3": { + "project_relative_path": "docs/feat-jev-combined/evidence/laya-repeated-20260921/round3-typed-fullinput/predictions.jsonl", + "sha256": "36c637ed48fbcb9eba34494c73bab23d7a539b745b63beee901d8736abfff450" + } + }, + "record_audit": [ + { + "record_id": "boolq/11b00c8dcb82dfbdaf02aa139e8c70603fb8c432e0fc51873ca00b20815a90f5", + "group_id": "evidence-group-83d16ec1b2a65c49a953cfc8b850422af8beb2758d72f39b23cf3e022be597a1", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/155bc4f5246c5b35ec14d7a2f697381de04c2cefbbc293b387bfdf83f2fea269", + "group_id": "evidence-group-9984381b97b26a3aa599aab8033350847b8b1cd7f6e2ed31ac1f8e954228e734", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/179184609853d9f8f1f862bd5c3113e363f622606fcf5c188cf0da1d3ed61e86", + "group_id": "evidence-group-afe52b84e10e464fe71423d120372724c1bb668d7dc611953a6946c3968c9b29", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/31fa2124dc5769d9e6fae53be4190290131f8fb025a089f8372d6f5d61504af5", + "group_id": "evidence-group-47ad7ea2eb35ffe5d68eebc068d2c2dc4668d58316bc11a90870609bd735fa77", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/3f87d9cb171dfe78b389d22677512f92cb48f7eedeefd35f2db1a2444cf08af2", + "group_id": "evidence-group-0ffadfb3889de88a9be6abacda9f33200265f55f988f8f4e30d679a6abb60220", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/42a6042685586d3796f9fe4460dd6e9cec208f39de59f765dcbef491fe576361", + "group_id": "evidence-group-23bbdb536073ab9fc895075ee72551bfc55632aaaec7a744418f9e8ada9c6993", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/46e903566fe8817ff37dce61d1ef85b757aa465dadb873c6356b6fc23f58476b", + "group_id": "evidence-group-0598f311fcee4030bb3e055fd8b809d3932df9668305a3243086d7fdb94ea925", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/4c460ed013de1b5d82c629bbf4e3bdbdaf4441a81341404d48d8e02ead5b9814", + "group_id": "evidence-group-a3f2fee0d01c242524e9d213c55f9f3e9f719327de2336257b7a3aedff15bbb5", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "boolq/58b070021c1d992130604f309a01b95166a281de0499349346694570a6c6ac57", + "group_id": "evidence-group-7b4397539e19353c30f0bf4c65423110a9ac730c11d28cb2c752348d7a6e9e37", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/5ba6d06215b4f7b269d0b30369fea59d748f02f41d1977f14c46c9153cd45cde", + "group_id": "evidence-group-0e7a5e22992089e0acef38c45a6a3764fe0cd682a8985ecfd39eb1475f5f0a1c", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/6909b3bd9078683a97a30ee70c6f3e91a9b8848a34efc0784eddf12cfa305cd4", + "group_id": "evidence-group-ee790b93179b77e8db71445f78bd460a066b750e6f458780297e74943d7bd257", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/791b5a767a61e438fe080d99068ef294b361fead41dbfc61b61021e33f2fff9f", + "group_id": "evidence-group-f64dca5a95a0d138d7ab52fcaffec2b80805f192bc5b892ece8256f5dfc238f7", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/8703574dc0db27742840afd3c1429829ef2f81c557670018e6ad5456fc53b150", + "group_id": "evidence-group-5b314cc015efa123085a0ddd9ab13485d0ac7b6809e85a039e3c43e817e1fb3e", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/a1ea17311c3aea8a751a1cd170483c439ce1338b451927c7db07692878ff0d82", + "group_id": "evidence-group-d7abd460afbf15ef627bcaf52168fde0428e0431b694dfee9a13fa42c925b59e", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/cca90e3be9b421447fe9ae0594193ab92a92adb0083f27a46bc224ad0e67c46c", + "group_id": "evidence-group-9735dc5df481652207fbd054ff5dad0d4ce8e7ce6c039aebe80d1e3ca8c8ee5d", + "dataset": "boolq.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "boolq/dbadfab033f2646b1a933fad5e132dc65cd8998c337d0332763f1df07680d7e7", + "group_id": "evidence-group-8ab83ce8fefb5340c8f2629e2ee8c09039f965976c87796f8b2d6d58b40f3017", + "dataset": "boolq.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:0c48f870b5edcb7a8dcfeb58087c7e3e56a9a2d89a0687e66d9974557c74556a:love", + "group_id": "goemotions-content:0c48f870b5edcb7a8dcfeb58087c7e3e56a9a2d89a0687e66d9974557c74556a", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:0d6f68c6aab93172edb30824a77a2f20e857a379845cfc16d8315af53d05bc1f:approval", + "group_id": "goemotions-content:0d6f68c6aab93172edb30824a77a2f20e857a379845cfc16d8315af53d05bc1f", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "goemotions:1285f8257d6a85acd0b9b01bfbebadff45ec085608b755eb415c74c624f8f5eb:optimism", + "group_id": "goemotions-content:1285f8257d6a85acd0b9b01bfbebadff45ec085608b755eb415c74c624f8f5eb", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:30e748ace2950567f2e1dcc3c638f6d5fc1e2df8078792928acf9550168933c7:approval", + "group_id": "goemotions-content:30e748ace2950567f2e1dcc3c638f6d5fc1e2df8078792928acf9550168933c7", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "goemotions:3b0e3210dddfd4dbb4b0e76cf62299246fd5a912229994a9e07de7dfc64c2262:embarrassment", + "group_id": "goemotions-content:3b0e3210dddfd4dbb4b0e76cf62299246fd5a912229994a9e07de7dfc64c2262", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:4308985a124eb5964a14e35ae0d7db83da5f33c019222497141a9a455cb5500c:joy", + "group_id": "goemotions-content:4308985a124eb5964a14e35ae0d7db83da5f33c019222497141a9a455cb5500c", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:43ef087b074a832ce71850c4c8d1672ca1d68b8b03830f65be8a52da990b454c:surprise", + "group_id": "goemotions-content:43ef087b074a832ce71850c4c8d1672ca1d68b8b03830f65be8a52da990b454c", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:4c893be6bdf7ad879db4d2299867417db381418ebeed081d85cbd13e6b3371a6:anger", + "group_id": "goemotions-content:4c893be6bdf7ad879db4d2299867417db381418ebeed081d85cbd13e6b3371a6", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:5695bd603bf8ce60300298693b29b4295f2d34c083c7e52bff826db80a6ea996:grief", + "group_id": "goemotions-content:5695bd603bf8ce60300298693b29b4295f2d34c083c7e52bff826db80a6ea996", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:5bb2f0355125b6ea32d14bc489651edeb09f6b4ef9f51b90b8ed330b8fd14cef:sadness", + "group_id": "goemotions-content:5bb2f0355125b6ea32d14bc489651edeb09f6b4ef9f51b90b8ed330b8fd14cef", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:8285d89a0201a60100cf7540b9e92fbf0564e50afc4e58afb812f91b6fc64cba:desire", + "group_id": "goemotions-content:8285d89a0201a60100cf7540b9e92fbf0564e50afc4e58afb812f91b6fc64cba", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:a312dda34fe450feab6e22e6d2aa7493d77c757d4800e2bfd9904549ff4c3756:realization", + "group_id": "goemotions-content:a312dda34fe450feab6e22e6d2aa7493d77c757d4800e2bfd9904549ff4c3756", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:cfa49c3d3a46c50e494455f5f1aa94c401a83a9450181bcf5d1f64ae75e30e64:annoyance", + "group_id": "goemotions-content:cfa49c3d3a46c50e494455f5f1aa94c401a83a9450181bcf5d1f64ae75e30e64", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "yes", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:e97ef69f22521e136e92eee4bbc1e46d8571bf91280b2bc7ae24f2e1ec5d0745:confusion", + "group_id": "goemotions-content:e97ef69f22521e136e92eee4bbc1e46d8571bf91280b2bc7ae24f2e1ec5d0745", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:e97ef69f22521e136e92eee4bbc1e46d8571bf91280b2bc7ae24f2e1ec5d0745:desire", + "group_id": "goemotions-content:e97ef69f22521e136e92eee4bbc1e46d8571bf91280b2bc7ae24f2e1ec5d0745", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "yes", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": false, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "goemotions:ee392e883c467546e7e603f5e6b94702211b3b4af3519a0bb7bd3b21bf4cbfd9:joy", + "group_id": "goemotions-content:ee392e883c467546e7e603f5e6b94702211b3b4af3519a0bb7bd3b21bf4cbfd9", + "dataset": "score.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0000105", + "group_id": "hs3-context-7a2fa0f0fe7894bf1abb41d8518e6962c162827ae154f3d80b78717be95835b5", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0001238", + "group_id": "hs3-context-6b5a52f42497fe8358c3d2ff591c263110d309fea01ff60990884ea43b9fcda2", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0001994", + "group_id": "hs3-context-467315e7063b6cb4c0b2914bd11e5be08b34591c4b5b7a50b6b1cbdaacf67c5e", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0008261", + "group_id": "hs3-context-0ea043fe9da180d611ed606527d5ec5d2aa4b9fd01799d01cc1a7a188592243f", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0010645", + "group_id": "hs3-context-e006169b0c265c71fa58ed56f685920a1b7e7e46bc2c6aeb71e68f9e72760e8b", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "no", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0010788", + "group_id": "hs3-context-6c870e63acf9c7c0026b23a8adc69509642da7798f11b1d3efcd7396b04c4885", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "no" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0011197", + "group_id": "hs3-context-58d30fb5fab1cfda3286562df2aaa1b9b492103c730d266a26acabe8067e2254", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0014165", + "group_id": "hs3-context-b086e3a56a3f29ef420f7a11867567e2464e6d4771744797ba3c9cb4f60075f2", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0014712", + "group_id": "hs3-context-b7bdb036aaa37153527cb2232c2d31505b414607b3dfe4b017edafbb7ece826f", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0015735", + "group_id": "hs3-context-b059df1a830f6f3f600339976f2000210e7df5eb69e99d00ebb5db5f0a0a3394", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0017850", + "group_id": "hs3-context-79018d8f60c4146ca7266b4178fd5c5dd283ccfa1b2022891749ae089273259f", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0018442", + "group_id": "hs3-context-b6709abe80497564f903e440eb8594cc8c589156807d75b5fcde781d2d390841", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "no", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0019640", + "group_id": "hs3-context-dd4926767c644b6b2061bb3ef3e4662e588bd2870aa2d19e3793ca520e92da3b", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0020714", + "group_id": "hs3-context-fa5f268e4c1958de8ee63a7ae4bef26c50d3d4e1d5bed178b6098657b84c20eb", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "yes", + "APUS-OpenJev 9B": "yes", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0024669", + "group_id": "hs3-context-6ccc2824fad3e248830bdf04072734f1764cdf1f193fa7946ee423e4d153a860", + "dataset": "hs3.dev.records.jsonl", + "gold": "yes", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "yes", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "hs3-f6d145777bcbde96137596340fab89793acd1031-train-0032390", + "group_id": "hs3-context-cabcd8323a4a475b9f0242c93ff3c050b8a00265d87f066d65af6d1493cf3208", + "dataset": "hs3.dev.records.jsonl", + "gold": "no", + "predictions": { + "APUS-OpenJev 4B": "no", + "APUS-OpenJev 9B": "no", + "Jev official API": "no", + "Laya typed / full input": "yes" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:0245df99-2909-465a-861e-7fbca948e82f:dc1847f7-919b-4a2f-b778-2ee33edacc46", + "group_id": "mind2web-episode:0245df99-2909-465a-861e-7fbca948e82f", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-a1a3ad1a8495aa678cbe", + "predictions": { + "APUS-OpenJev 4B": "candidate-a1a3ad1a8495aa678cbe", + "APUS-OpenJev 9B": "candidate-a1a3ad1a8495aa678cbe", + "Jev official API": "candidate-a1a3ad1a8495aa678cbe", + "Laya typed / full input": "candidate-a1a3ad1a8495aa678cbe" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mind2web:0b59dd33-7f6a-48df-aa1e-9cc67177287f:b99c5581-8a56-4bd7-bbe4-782795ebf93c", + "group_id": "mind2web-episode:0b59dd33-7f6a-48df-aa1e-9cc67177287f", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-9409b72a36db8d4a0a79", + "predictions": { + "APUS-OpenJev 4B": "candidate-9409b72a36db8d4a0a79", + "APUS-OpenJev 9B": "candidate-9409b72a36db8d4a0a79", + "Jev official API": "candidate-9409b72a36db8d4a0a79", + "Laya typed / full input": "candidate-9409b72a36db8d4a0a79" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mind2web:1efdcd9d-ebc6-4bb7-8823-e54dfe25f409:f6e611c9-ad21-49ca-a841-7ad529b56c95", + "group_id": "mind2web-episode:1efdcd9d-ebc6-4bb7-8823-e54dfe25f409", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-fb3025ec54721504f9b5", + "predictions": { + "APUS-OpenJev 4B": "candidate-fb3025ec54721504f9b5", + "APUS-OpenJev 9B": "candidate-fb3025ec54721504f9b5", + "Jev official API": "candidate-fb3025ec54721504f9b5", + "Laya typed / full input": "candidate-fb3025ec54721504f9b5" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mind2web:274571ea-fc2f-4353-86ba-00ecb112d6d2:e6345ea9-a5a4-4b88-95b5-4efececed261", + "group_id": "mind2web-episode:274571ea-fc2f-4353-86ba-00ecb112d6d2", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-3b25ae8d064a298ef021", + "predictions": { + "APUS-OpenJev 4B": "candidate-3b25ae8d064a298ef021", + "APUS-OpenJev 9B": "candidate-3b25ae8d064a298ef021", + "Jev official API": "candidate-3b25ae8d064a298ef021", + "Laya typed / full input": "candidate-9827a0d3bbbb4c3f2ef3" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:3f0988e0-e3a4-4dd7-a89f-482175a474a8:a684c98a-c238-4bef-b2ad-f476f07d73f8", + "group_id": "mind2web-episode:3f0988e0-e3a4-4dd7-a89f-482175a474a8", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-016db9380de74cbcab5d", + "predictions": { + "APUS-OpenJev 4B": "candidate-016db9380de74cbcab5d", + "APUS-OpenJev 9B": "candidate-016db9380de74cbcab5d", + "Jev official API": "candidate-6ff913a423628d92b636", + "Laya typed / full input": "candidate-ae8922f024c12c266278" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:44a12ff5-0172-444a-b979-f224162c1aa8:5b8da6f5-c53c-4b69-bfad-7bdfd2e6ce20", + "group_id": "mind2web-episode:44a12ff5-0172-444a-b979-f224162c1aa8", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-3ef3bb624316b6eb3508", + "predictions": { + "APUS-OpenJev 4B": "candidate-94dddd06eb43ed2921b0", + "APUS-OpenJev 9B": "candidate-f5c1b8b97d7add30b043", + "Jev official API": "candidate-94dddd06eb43ed2921b0", + "Laya typed / full input": "candidate-331b8cbdf48fd7f61188" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:5b56d5b8-1f41-43ca-9f21-369d849f1aa0:edb09eb7-6a8c-4aeb-9b52-796762ca821d", + "group_id": "mind2web-episode:5b56d5b8-1f41-43ca-9f21-369d849f1aa0", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-2dc16f49fe3acb051b39", + "predictions": { + "APUS-OpenJev 4B": "candidate-2dc16f49fe3acb051b39", + "APUS-OpenJev 9B": "candidate-2dc16f49fe3acb051b39", + "Jev official API": "candidate-2dc16f49fe3acb051b39", + "Laya typed / full input": "candidate-ea45663120f48fa54b88" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:5f9182dc-d35d-4c0e-9abe-cd913c136528:8450e2c6-f8aa-40bc-876e-21cf29a8cb77", + "group_id": "mind2web-episode:5f9182dc-d35d-4c0e-9abe-cd913c136528", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-d2449a13158601200ec0", + "predictions": { + "APUS-OpenJev 4B": "candidate-d2449a13158601200ec0", + "APUS-OpenJev 9B": "candidate-d2449a13158601200ec0", + "Jev official API": "candidate-33ff2e219977f0035115", + "Laya typed / full input": "candidate-bfd1999ae1c4a3ad89f3" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:6e565708-43e2-492b-9f1d-25d51387dcf7:bee75aa3-6f7d-4626-be6f-1b217ac16733", + "group_id": "mind2web-episode:6e565708-43e2-492b-9f1d-25d51387dcf7", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-bb1945d8d438379625ee", + "predictions": { + "APUS-OpenJev 4B": "candidate-bb1945d8d438379625ee", + "APUS-OpenJev 9B": "candidate-0fb9a6cde602f212d85e", + "Jev official API": "candidate-0fb9a6cde602f212d85e", + "Laya typed / full input": "candidate-69abfa80e562043e1fa3" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:80e12375-19ad-400f-9e35-2a3853173bed:e89bb795-2d24-4e2c-bcae-1294e3501dfa", + "group_id": "mind2web-episode:80e12375-19ad-400f-9e35-2a3853173bed", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-41c5af16d7bf373cc74c", + "predictions": { + "APUS-OpenJev 4B": "candidate-41c5af16d7bf373cc74c", + "APUS-OpenJev 9B": "candidate-41c5af16d7bf373cc74c", + "Jev official API": "candidate-41c5af16d7bf373cc74c", + "Laya typed / full input": "candidate-41c5af16d7bf373cc74c" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mind2web:8368b990-c6ca-4cfe-a7ab-c2a88697639d:bf14f1d4-470f-4110-b3f4-019a9f7d0aed", + "group_id": "mind2web-episode:8368b990-c6ca-4cfe-a7ab-c2a88697639d", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-8b6280aa9682e1a79ce0", + "predictions": { + "APUS-OpenJev 4B": "candidate-da095abe0a63c03f3114", + "APUS-OpenJev 9B": "candidate-e76caeae9d53c8942ed1", + "Jev official API": "candidate-da095abe0a63c03f3114", + "Laya typed / full input": "candidate-f8c1512ffc92c70eafe7" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:8eae88ef-9641-43c6-be6d-f8abc96d99fa:282c09d0-c9e0-4007-b88a-27887fe1e388", + "group_id": "mind2web-episode:8eae88ef-9641-43c6-be6d-f8abc96d99fa", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-976c2731e7518a1f1b2b", + "predictions": { + "APUS-OpenJev 4B": "candidate-976c2731e7518a1f1b2b", + "APUS-OpenJev 9B": "candidate-976c2731e7518a1f1b2b", + "Jev official API": "candidate-976c2731e7518a1f1b2b", + "Laya typed / full input": "candidate-7e020b58dbe771cceb6b" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:bef473f1-82a1-4359-a2c0-59b6dc2f6abb:22ad3562-e0f4-42c3-b096-8c173a47673c", + "group_id": "mind2web-episode:bef473f1-82a1-4359-a2c0-59b6dc2f6abb", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-c59cf6b3cdd1ee8dfe92", + "predictions": { + "APUS-OpenJev 4B": "candidate-c59cf6b3cdd1ee8dfe92", + "APUS-OpenJev 9B": "candidate-c59cf6b3cdd1ee8dfe92", + "Jev official API": "candidate-c59cf6b3cdd1ee8dfe92", + "Laya typed / full input": "candidate-7111d2a0c81d7ac049be" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:caafd610-202e-49d2-85d1-3f167f3ab443:d7bfb473-8c73-4808-96bc-187d00be5ad7", + "group_id": "mind2web-episode:caafd610-202e-49d2-85d1-3f167f3ab443", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-b56ad12ceea4c52feb3f", + "predictions": { + "APUS-OpenJev 4B": "candidate-b56ad12ceea4c52feb3f", + "APUS-OpenJev 9B": "candidate-b56ad12ceea4c52feb3f", + "Jev official API": "candidate-b56ad12ceea4c52feb3f", + "Laya typed / full input": "candidate-9ebb992dd5a329ac8366" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:d1de3d1a-3df1-4421-98f3-f8d078752893:e1f18ee3-1577-44fb-a283-1be215e5ae52", + "group_id": "mind2web-episode:d1de3d1a-3df1-4421-98f3-f8d078752893", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-888bb5997d1aba7b25f7", + "predictions": { + "APUS-OpenJev 4B": "candidate-888bb5997d1aba7b25f7", + "APUS-OpenJev 9B": "candidate-888bb5997d1aba7b25f7", + "Jev official API": "candidate-888bb5997d1aba7b25f7", + "Laya typed / full input": "candidate-0196e00c3795624bcc41" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mind2web:d637c171-dc6e-4a4e-a162-9c230e822932:9053f0e2-da05-4721-87b5-13edf923052b", + "group_id": "mind2web-episode:d637c171-dc6e-4a4e-a162-9c230e822932", + "dataset": "browser.dev.records.jsonl", + "gold": "candidate-34e89f28b090ce5327b0", + "predictions": { + "APUS-OpenJev 4B": "candidate-34e89f28b090ce5327b0", + "APUS-OpenJev 9B": "candidate-34e89f28b090ce5327b0", + "Jev official API": "candidate-34e89f28b090ce5327b0", + "Laya typed / full input": "candidate-d82967df7ccb99323f3d" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mnli/13f1e11494916138e91023e5bf4d2483ab5d93f00e2e1e558c5f595ca0e8e22d", + "group_id": "evidence-group-1ae4104ec56e749db2b0bfb640e71f29c6e29d797a0b7eefc4c77832f7de480e", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "neutral", + "Laya typed / full input": "neutral" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/3d8bd065ebff13c30fa3aa4943c5f2851ca8a5860bae22e25f212be97790ab28", + "group_id": "evidence-group-06e276f7577733f351ea270929ab121557f5ba0b4b7a7ffc8599a8dc2e7ca393", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/4bd800d8bc83435b531bb472e9078bb054a7bb68bf31036bc12502e177dea19b", + "group_id": "evidence-group-fc2f08eeeca529c675ba6a68c08c292b57689624394977ddbb1c731d34a4a7e8", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/4dc433ef7c48f1158e61aa4f1b568bac7ee2ede91001c9b8d933fd5c54e71820", + "group_id": "evidence-group-380981d6890572077b0d6b902aa0469166c29c7a700575399cd2fc7a8c078114", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/533107d0478441289d303edb55e3769ea75e77ee657738576b1e86f9f523ff09", + "group_id": "evidence-group-7d41c94d5aacee4210e9af39c38dded413bb0ed0e66b4ab138cb0cf9acc9d175", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "contradiction", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "contradiction" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mnli/53ac0a380403b0ccd3ae48ea7763a52664f1d1744f6244c25d5980d69fcf7f0f", + "group_id": "evidence-group-4743dba6976b2918897a2f0f6d686fdaccca4cb6735d4db706e4aced85eb44e2", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/5c5d3ecddeb08dada33aaa9e323aa10bdfa371aa2f535b2cd5f4f653b0e25e9c", + "group_id": "evidence-group-0293cb3342154c4098832723d3775fb7f1a88b31e6126f9db882ed1a0d8b491a", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/834677de1c970abc266e34006f2e681c7301ff5d043796d92da66364b792f235", + "group_id": "evidence-group-ccf7a781e6c7f6802cdc3b42bb6bad795f0ba0ae4ae9b10b286d9699ad373d3e", + "dataset": "mnli.dev.records.jsonl", + "gold": "contradiction", + "predictions": { + "APUS-OpenJev 4B": "contradiction", + "APUS-OpenJev 9B": "contradiction", + "Jev official API": "contradiction", + "Laya typed / full input": "contradiction" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/8f4dbf064e95ffddc782d0d50ab5e6e0fb83aab546d48b00bff5f03706183bf9", + "group_id": "evidence-group-cad5bd59c81306dbfbc2992418db0c5720f237fc174da87cd7360ead0af20f15", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "neutral", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mnli/9cf51bc8717b59b904d198884ba5b832b53cdc96a5b1aafee7203dca9bfc824d", + "group_id": "evidence-group-12003df0e147aae371c07114820d8e1271ea9e77fc426c2dfcd7ea5095216b12", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "neutral", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/ab466125416d6bc3f307f7fd25b78e63a8d90f8f54713027fdc7d75fffd8558d", + "group_id": "evidence-group-1d662b8bfb238ffbb5674491c34bbdb350bd550b8d944cfdf8c1f6e3d323c69a", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "neutral", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": false, + "Jev official API": false, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/b0220f9d099b621bde205ea6efa72624cc1948cceb2502cdc38d19c3762c2ba0", + "group_id": "evidence-group-60d29399f85535a7eef633c31f9819b7eec6d7a9800097bf817edabf2f4314f8", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "contradiction", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": false, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mnli/c8c695b22a0a846dde5f0ae3c17127f27c53f23219de2f4c97f2a9bebe1e0f81", + "group_id": "evidence-group-877e43849360055fc2c92bc7b5a7ee30b94048287fb4100ce4eb9004cd95610f", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/dba1b74655afea3f9676c7a8687e80c0da196b11f2324d6e3a41c3870b9d8591", + "group_id": "evidence-group-c6fb07e356610f631e4b8e0884fb1cef27da866d3a1286508ad84c9a1ad50eeb", + "dataset": "mnli.dev.records.jsonl", + "gold": "entailment", + "predictions": { + "APUS-OpenJev 4B": "entailment", + "APUS-OpenJev 9B": "entailment", + "Jev official API": "entailment", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": true + } + }, + { + "record_id": "mnli/e6856d10a89f46cbfc7fa77bf9266c482f130090a961f5bcb10be540e7c5fc2c", + "group_id": "evidence-group-833d5b5523f6e081b1f65c924bed3ba030ffbfef18baba106cb0c7bce86b930c", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "neutral", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + }, + { + "record_id": "mnli/e6cf83f056ba7587936d8026ff9709bc50de62e0dfb282f6b8d6aa85649e688d", + "group_id": "evidence-group-cd8d19e8d6333678e561021d9ffa0dbfb0b7047f3fe684fd553208ed862407a6", + "dataset": "mnli.dev.records.jsonl", + "gold": "neutral", + "predictions": { + "APUS-OpenJev 4B": "neutral", + "APUS-OpenJev 9B": "neutral", + "Jev official API": "neutral", + "Laya typed / full input": "entailment" + }, + "correct": { + "APUS-OpenJev 4B": true, + "APUS-OpenJev 9B": true, + "Jev official API": true, + "Laya typed / full input": false + } + } + ], + "limitations": [ + "Each task has only 16 questions; one answer changes task accuracy by 6.25 percentage points.", + "Development panel was reused during development and model selection; not evidence of general superiority.", + "Browser measures static candidate selection, not live website task completion.", + "Score includes no positive labels; an always-No predictor scores 16/16.", + "Laya uses the typed configuration with full-input formatting; three process repeats do not increase the independent sample size." + ] +}