tonywu71 commited on
Commit
b043d83
·
0 Parent(s):

feat: upload neomme files

Browse files
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
1_Pooling/config.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "embedding_dimension": 1792,
3
+ "pooling_mode": "mean",
4
+ "include_prompt": true
5
+ }
2_Normalize/config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "module_input_name": "sentence_embedding",
3
+ "module_output_name": "sentence_embedding"
4
+ }
LICENSE ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or Derivative
95
+ Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright 2026 H Company (hcompany.ai)
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
README.md ADDED
@@ -0,0 +1,125 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: sentence-transformers
3
+ license: apache-2.0
4
+ language:
5
+ - multilingual
6
+ base_model:
7
+ - Hcompany/NeoMME-800M-Retriever
8
+ pipeline_tag: sentence-similarity
9
+ tags:
10
+ - multimodal
11
+ - document-retrieval
12
+ - dense-retrieval
13
+ ---
14
+
15
+ <p align="left">
16
+ <img src="https://github.com/tonywu71/colpali-cookbooks/blob/6ef1332da6bcb48c7ef1f19b25bfa555be7031a8/assets/neomme/neomme_logo.webp?raw=true" alt="NeoMME logo" style="max-height: 140px;">
17
+ </p>
18
+
19
+ # NeoMME-Retriever (800M): Single-Tower Multimodal-Native Multilingual Foundation Encoder 🔎
20
+
21
+ > [!IMPORTANT]
22
+ > NeoMME-Retriever (800M) variants:
23
+ >
24
+ > - [Default (`transformers`)](https://huggingface.co/Hcompany/NeoMME-800M-Retriever): Returns dense and multi-vector embeddings together with a single forward pass. Recommended for most use cases and inference.
25
+ > - [ST dense](https://huggingface.co/Hcompany/NeoMME-800M-Retriever-ST-dense) [current]: Supports independent dense fine-tuning with Sentence Transformers.
26
+ > - [ST late-interaction](https://huggingface.co/Hcompany/NeoMME-800M-Retriever-ST-late): Supports independent multi-vector fine-tuning with Sentence Transformers.
27
+
28
+ [![Hugging Face](https://img.shields.io/badge/Model_doc-FFD21E?style=for-the-badge&logo=huggingface&logoColor=000)](https://huggingface.co/docs/transformers/en/model_doc/neomme)
29
+ [![Hugging Face](https://img.shields.io/badge/Collection-FFD21E?style=for-the-badge&logo=huggingface&logoColor=000)](https://hf.co/collections/Hcompany/neomme)
30
+ [![arXiv](https://img.shields.io/badge/arXiv-coming_soon-b31b1b.svg?style=for-the-badge)](https://arxiv.org)
31
+
32
+ NeoMME-800M-Retriever-ST-dense is a model for multimodal document retrieval. Fine-tuned from [NeoMME-800M](https://huggingface.co/Hcompany/NeoMME-800M), it encodes text queries and documents (text or page screenshots) using one shared bidirectional Transformer encoder.
33
+
34
+ This model can be used with Sentence Transformers, but can only generate dense embeddings.
35
+
36
+ <table>
37
+ <thead>
38
+ <tr style="background-color: rgba(146, 81, 247, 0.20);"><th>Specification</th><th>Value</th></tr>
39
+ </thead>
40
+ <tbody>
41
+ <tr><td>Parameters</td><td>800M</td></tr>
42
+ <tr><td>Vocabulary</td><td>131,072 tokens</td></tr>
43
+ <tr><td>Context length</td><td>16,384 tokens</td></tr>
44
+ <tr><td>Hidden size</td><td>1,792</td></tr>
45
+ <tr><td>Image patches</td><td>32 × 32 pixels, up to 2,048 pixels on the longest side (default)</td></tr>
46
+ <tr><td>Dense embeddings</td><td>1,792 dimensions (Matryoshka: [128, 256, 512, 1,024, 1,792])</td></tr>
47
+ <tr><td>Dense pooling strategy</td><td>Mean</td></tr>
48
+ </tbody>
49
+ </table>
50
+
51
+ Dense embeddings are L2-normalized and use cosine similarity. They match `NeoMMEForRetrieval.dense_embeddings`.
52
+
53
+ ## Performance
54
+
55
+ All scores use the metric shown at the full trained dimensions. Higher is better. ViDoRe v3, v2, and v1 measure visual document retrieval, while BEIR-15 measures text retrieval.
56
+
57
+ <table>
58
+ <thead>
59
+ <tr><th rowspan="2">Benchmark</th><th rowspan="2">Metric</th><th colspan="2">NeoMME-260M</th><th colspan="2" style="background-color: rgba(37, 99, 235, 0.20);">NeoMME-800M</th></tr>
60
+ <tr><th>Late interaction</th><th>Dense</th><th>Late interaction</th><th style="background-color: rgba(37, 99, 235, 0.20);">Dense [current]</th></tr>
61
+ </thead>
62
+ <tbody>
63
+ <tr><td>ViDoRe v3</td><td>nDCG@10</td><td>0.5226</td><td>0.3907</td><td>0.5560</td><td style="background-color: rgba(37, 99, 235, 0.20);">0.4391</td></tr>
64
+ <tr><td>ViDoRe v2</td><td>nDCG@5</td><td>0.5218</td><td>0.4075</td><td>0.5591</td><td style="background-color: rgba(37, 99, 235, 0.20);">0.4475</td></tr>
65
+ <tr><td>ViDoRe v1</td><td>nDCG@5</td><td>0.8598</td><td>0.7552</td><td>0.8744</td><td style="background-color: rgba(37, 99, 235, 0.20);">0.7993</td></tr>
66
+ <tr><td>BEIR-15</td><td>nDCG@10</td><td>0.4881</td><td>0.3055</td><td>0.5126</td><td style="background-color: rgba(37, 99, 235, 0.20);">0.3686</td></tr>
67
+ </tbody>
68
+ </table>
69
+
70
+
71
+ ## Usage
72
+
73
+ ```bash
74
+ pip install -U "sentence-transformers[image]"
75
+ ```
76
+
77
+ ```python
78
+ from sentence_transformers import SentenceTransformer
79
+
80
+
81
+ model = SentenceTransformer("Hcompany/NeoMME-800M-Retriever-ST-dense")
82
+
83
+ queries = [
84
+ "Quelle partie de la production pétrolière du Kazakhstan provient de champs en mer ?",
85
+ "Which hour of the day had the highest overall electricity generation in 2019?",
86
+ ]
87
+ documents = [
88
+ "https://github.com/tonywu71/colpali-cookbooks/blob/main/examples/data/shift_kazakhstan.jpg?raw=true",
89
+ "https://github.com/tonywu71/colpali-cookbooks/blob/main/examples/data/energy_electricity_generation.jpg?raw=true",
90
+ ]
91
+
92
+ query_embeddings = model.encode_query(queries, convert_to_tensor=True)
93
+ document_embeddings = model.encode_document(documents, convert_to_tensor=True)
94
+ scores = model.similarity(query_embeddings, document_embeddings)
95
+
96
+ # Expected: scores[0, 0] > scores[0, 1] and scores[1, 1] > scores[1, 0].
97
+ print(scores)
98
+ ```
99
+
100
+ The score tensor has shape `(num_queries, num_documents)` and `scores[i, j]` is the score between query `i` and document `j`. A larger value indicates a closer match.
101
+
102
+ ## Training
103
+
104
+ NeoMME-800M-Retriever was fine-tuned from [NeoMME-800M](https://huggingface.co/Hcompany/NeoMME-800M) on text retrieval and document-page images. Training uses a joint late-interaction and Matryoshka dense contrastive objective.
105
+
106
+ The NeoMME technical report describes the full fine-tuning recipe (will be released soon).
107
+
108
+ ## Limitations
109
+
110
+ With Sentence Transformers, only one of the two retrieval heads can be used at a time.
111
+
112
+ ## License
113
+
114
+ Model weights are released under the Apache 2.0 license.
115
+
116
+ ## Citation
117
+
118
+ ```bibtex
119
+ @techreport{neomme2026,
120
+ title = {NeoMME: A Single-Tower Multimodal-Native Multilingual Foundation Encoder for Efficient Fine-Tuning and Inference},
121
+ author = {Lac, Aurélien and Wu, Tony},
122
+ institution = {H Company},
123
+ year = {2026}
124
+ }
125
+ ```
chat_template.jinja ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {#-
2
+ NeoMME retrieval prompt template. Not a conversation format: NeoMME is an encoder and has no chat turns.
3
+
4
+ `task` selects the retrieval side. A sentence-transformers v6 `MultiVectorEncoder` routes
5
+ `task="query"` / `task="document"` here when the template declares the variable.
6
+
7
+ task="query" -> <query> … N x <mask> ColBERT-style learned query expansion. <mask> is the
8
+ masked-diffusion fill token. The exporter sets N from the
9
+ checkpoint.
10
+ task="document" -> <doc> … Text passages. Image pages take the same <doc> prefix
11
+ followed by one image token placeholder, which the processor
12
+ expands after it knows the patch grid.
13
+ -#}
14
+ {%- if task is not defined -%}
15
+ {{- raise_exception("NeoMME chat templates require task='query' or task='document'.") -}}
16
+ {%- endif -%}
17
+ {%- if task not in ['query', 'document'] -%}
18
+ {{- raise_exception("task=" ~ task ~ " is not supported: expected 'query' or 'document'.") -}}
19
+ {%- endif -%}
20
+ {%- if messages is not defined or not messages -%}
21
+ {{- raise_exception("NeoMME chat conversations must contain at least one message.") -}}
22
+ {%- endif -%}
23
+
24
+ {# Collect content and validate one retrieval item without imposing batch-level policy. #}
25
+ {%- set state = namespace(text='', has_text=false, image_count=0) -%}
26
+ {%- for message in messages -%}
27
+ {%- set content = message.content -%}
28
+ {%- set items = [{'type': 'text', 'text': content}] if content is string else content -%}
29
+ {%- for item in items -%}
30
+ {%- if item.type == 'text' -%}
31
+ {%- if image_token in item.text -%}
32
+ {{- raise_exception(image_token ~ " is reserved for image documents.") -}}
33
+ {%- endif -%}
34
+ {%- set state.has_text = true -%}
35
+ {%- set state.text = state.text + item.text -%}
36
+ {%- elif item.type == 'image' -%}
37
+ {%- if item.image is not defined or item.image is none or item.image == '' -%}
38
+ {{- raise_exception("NeoMME image content must provide an image source.") -}}
39
+ {%- endif -%}
40
+ {%- set state.image_count = state.image_count + 1 -%}
41
+ {%- elif item.type == 'image_url' -%}
42
+ {%- if item.image_url is not defined or not item.image_url -%}
43
+ {{- raise_exception("NeoMME image_url content must provide an image source.") -}}
44
+ {%- endif -%}
45
+ {%- set state.image_count = state.image_count + 1 -%}
46
+ {%- else -%}
47
+ {{- raise_exception("NeoMME chat templates do not support content type " ~ item.type ~ ".") -}}
48
+ {%- endif -%}
49
+ {%- endfor -%}
50
+ {%- endfor -%}
51
+
52
+ {%- if state.image_count and state.has_text -%}
53
+ {{- raise_exception("NeoMME cannot encode text and images in the same conversation.") -}}
54
+ {%- endif -%}
55
+ {%- if state.image_count > 1 -%}
56
+ {{- raise_exception("NeoMME accepts one image document per conversation.") -}}
57
+ {%- endif -%}
58
+ {%- if state.image_count and task != 'document' -%}
59
+ {{- raise_exception("NeoMME image content must use task='document'.") -}}
60
+ {%- endif -%}
61
+
62
+ {%- set content = image_token if state.image_count else state.text -%}
63
+ {%- if task == 'query' -%}
64
+ {{- query_token + content + mask_token * 10 -}}
65
+ {%- else -%}
66
+ {{- document_token + content -}}
67
+ {%- endif -%}
config.json ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "NeoMMEModel"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "document_token_id": 5,
8
+ "dtype": "bfloat16",
9
+ "embedding_dim": 128,
10
+ "embedding_rank": 256,
11
+ "head_dim": 64,
12
+ "hidden_act": "relu2",
13
+ "hidden_size": 1792,
14
+ "image_token_id": 6,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 6400,
17
+ "layer_types": [
18
+ "sliding_attention",
19
+ "sliding_attention",
20
+ "sliding_attention",
21
+ "sliding_attention",
22
+ "sliding_attention",
23
+ "full_attention",
24
+ "sliding_attention",
25
+ "sliding_attention",
26
+ "sliding_attention",
27
+ "sliding_attention",
28
+ "sliding_attention",
29
+ "full_attention",
30
+ "sliding_attention",
31
+ "sliding_attention",
32
+ "sliding_attention",
33
+ "sliding_attention",
34
+ "sliding_attention",
35
+ "full_attention",
36
+ "sliding_attention",
37
+ "full_attention"
38
+ ],
39
+ "max_position_embeddings": 16384,
40
+ "mlp_bias": false,
41
+ "model_type": "neomme",
42
+ "norm_eps": 1e-06,
43
+ "num_attention_heads": 28,
44
+ "num_hidden_layers": 20,
45
+ "num_key_value_heads": 7,
46
+ "pad_token_id": 0,
47
+ "patch_size": 32,
48
+ "per_layer_config": {
49
+ "01": {
50
+ "sliding_window": 1024
51
+ },
52
+ "03": {
53
+ "sliding_window": 1024
54
+ },
55
+ "05": {
56
+ "sliding_window": null
57
+ },
58
+ "06": {
59
+ "sliding_window": 1024
60
+ },
61
+ "08": {
62
+ "sliding_window": 1024
63
+ },
64
+ "10": {
65
+ "sliding_window": 1024
66
+ },
67
+ "11": {
68
+ "sliding_window": null
69
+ },
70
+ "13": {
71
+ "sliding_window": 1024
72
+ },
73
+ "15": {
74
+ "sliding_window": 1024
75
+ },
76
+ "17": {
77
+ "sliding_window": null
78
+ },
79
+ "18": {
80
+ "sliding_window": 1024
81
+ },
82
+ "19": {
83
+ "sliding_window": null
84
+ }
85
+ },
86
+ "residual_multiplier": 0.15811388300841897,
87
+ "rope_parameters": {
88
+ "full_attention": {
89
+ "partial_rotary_factor": 0.25,
90
+ "rope_theta": 1000000.0,
91
+ "rope_type": "default"
92
+ },
93
+ "sliding_attention": {
94
+ "partial_rotary_factor": 1.0,
95
+ "rope_theta": 10000.0,
96
+ "rope_type": "default"
97
+ }
98
+ },
99
+ "sliding_window": 256,
100
+ "tie_word_embeddings": true,
101
+ "transformers_version": "5.16.0.dev0",
102
+ "vocab_size": 131072
103
+ }
config_sentence_transformers.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "__version__": {
3
+ "sentence_transformers": "6.0.0"
4
+ },
5
+ "default_prompt_name": null,
6
+ "model_type": "SentenceTransformer",
7
+ "prompts": {
8
+ "document": "",
9
+ "query": ""
10
+ },
11
+ "similarity_fn_name": "cosine"
12
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:949454a191f1ed98e074be1e81317e89b24d29d7dc4d4567cbfff857c47ef81e
3
+ size 1587451080
modules.json ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "idx": 0,
4
+ "name": "0",
5
+ "path": "",
6
+ "type": "sentence_transformers.base.modules.transformer.Transformer"
7
+ },
8
+ {
9
+ "idx": 1,
10
+ "name": "1",
11
+ "path": "1_Pooling",
12
+ "type": "sentence_transformers.sentence_transformer.modules.pooling.Pooling"
13
+ },
14
+ {
15
+ "idx": 2,
16
+ "name": "2",
17
+ "path": "2_Normalize",
18
+ "type": "sentence_transformers.base.modules.normalize.Normalize"
19
+ }
20
+ ]
processor_config.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_processor": {
3
+ "do_convert_rgb": true,
4
+ "do_normalize": true,
5
+ "do_rescale": true,
6
+ "do_resize": true,
7
+ "image_mean": [
8
+ 0.5,
9
+ 0.5,
10
+ 0.5
11
+ ],
12
+ "image_processor_type": "NeoMMEImageProcessor",
13
+ "image_std": [
14
+ 0.5,
15
+ 0.5,
16
+ 0.5
17
+ ],
18
+ "max_side": 2048,
19
+ "patch_size": 32,
20
+ "resample": 2,
21
+ "rescale_factor": 0.00392156862745098
22
+ },
23
+ "processor_class": "NeoMMEProcessor"
24
+ }
sentence_bert_config.json ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "transformer_task": "feature-extraction",
3
+ "modality_config": {
4
+ "text": {
5
+ "method": "forward",
6
+ "method_output_name": "last_hidden_state"
7
+ },
8
+ "image": {
9
+ "method": "forward",
10
+ "method_output_name": "last_hidden_state"
11
+ },
12
+ "message": {
13
+ "method": "forward",
14
+ "method_output_name": "last_hidden_state",
15
+ "format": "structured"
16
+ }
17
+ },
18
+ "module_output_name": "token_embeddings",
19
+ "unpad_inputs": false
20
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "clean_up_tokenization_spaces": false,
4
+ "document_token": "<doc>",
5
+ "eos_token": "<eos>",
6
+ "extra_special_tokens": [],
7
+ "image_token": "<img>",
8
+ "mask_token": "<mask>",
9
+ "model_input_names": [
10
+ "input_ids",
11
+ "attention_mask"
12
+ ],
13
+ "model_max_length": 16384,
14
+ "pad_token": "<pad>",
15
+ "processor_class": "NeoMMEProcessor",
16
+ "query_token": "<query>",
17
+ "row_token": "<row>",
18
+ "tokenizer_class": "TokenizersBackend",
19
+ "truncation_side": "right",
20
+ "unk_token": "<unk>"
21
+ }