JarvisPei AlexDL commited on
Commit
b50e043
·
0 Parent(s):

Release NeoHorse-Jev-4B

Browse files

Co-authored-by: AlexDL <AlexDL@users.noreply.huggingface.co>

Files changed (45) hide show
  1. .gitattributes +38 -0
  2. DEPLOYMENT.md +398 -0
  3. README.md +584 -0
  4. SHA256SUMS +43 -0
  5. assets/.gitkeep +0 -0
  6. assets/jev-six-demo-grid.gif +3 -0
  7. assets/jev-snake-demo.gif +3 -0
  8. backbone/config.json +112 -0
  9. backbone/model-00001-of-00003.safetensors +3 -0
  10. backbone/model-00002-of-00003.safetensors +3 -0
  11. backbone/model-00003-of-00003.safetensors +3 -0
  12. backbone/model.safetensors.index.json +730 -0
  13. backbone/preprocessor_config.json +21 -0
  14. config.json +22 -0
  15. dist/neohorse_decision-1.0.0-py3-none-any.whl +0 -0
  16. environment.json +17 -0
  17. example_request.json +26 -0
  18. model_manifest.json +22 -0
  19. package/pyproject.toml +24 -0
  20. package/src/neohorse_decision/__init__.py +5 -0
  21. package/src/neohorse_decision/_inference.py +70 -0
  22. package/src/neohorse_decision/_vendor/LICENSE +203 -0
  23. package/src/neohorse_decision/_vendor/NOTICE.md +11 -0
  24. package/src/neohorse_decision/_vendor/__init__.py +1 -0
  25. package/src/neohorse_decision/_vendor/model.py +300 -0
  26. package/src/neohorse_decision/_vendor/schema.py +112 -0
  27. package/src/neohorse_decision/cli.py +25 -0
  28. package/src/neohorse_decision/engine.py +46 -0
  29. package/src/neohorse_decision/image_input.py +41 -0
  30. package/src/neohorse_decision/server.py +90 -0
  31. package/src/neohorse_decision/systemone.py +56 -0
  32. package/src/neohorse_decision/vision.py +75 -0
  33. pointer_head.safetensors +3 -0
  34. tokenizer/chat_template.jinja +154 -0
  35. tokenizer/tokenizer.json +3 -0
  36. tokenizer/tokenizer_config.json +32 -0
  37. vision/LICENSE +202 -0
  38. vision/README.md +28 -0
  39. vision/__init__.py +2 -0
  40. vision/base_vision_provenance.json +2795 -0
  41. vision/example.py +20 -0
  42. vision/example_request.json +11 -0
  43. vision/http_example.py +32 -0
  44. vision/predictor.py +2 -0
  45. vision/verification.json +130 -0
.gitattributes ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ assets/jev-snake-demo.gif filter=lfs diff=lfs merge=lfs -text
37
+ tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ assets/jev-six-demo-grid.gif filter=lfs diff=lfs merge=lfs -text
DEPLOYMENT.md ADDED
@@ -0,0 +1,398 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Deployment and API Guide
2
+
3
+ This guide covers installation and text and image inference with the `neohorse_decision` runtime. See the [README](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B) for the model overview and quickstart, or the [backend guide](https://github.com/TokenRhythm/NeoHorse/blob/main/jev/infer/README.md) for the separate vLLM and SGLang adapters.
4
+
5
+ ## 1. Installation
6
+
7
+ Download the complete release from the [Hugging Face model repository](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/tree/main) and install its bundled runtime. Loading the model requires `backbone/`, `tokenizer/`, `pointer_head.safetensors`, and `model_manifest.json`. Image examples are also included in the release.
8
+
9
+ The recorded environment is Linux, Python 3.12, PyTorch 2.8.0, Transformers 5.17.0, Triton 3.7.1, and flash-linear-attention 0.5.2, with a CUDA GPU that supports BF16. Version details are in [environment.json](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/environment.json). The following commands assume these ML dependencies are already installed in an isolated environment:
10
+
11
+ ```bash
12
+ export MODEL_DIR="/path/to/NeoHorse-Jev-4B"
13
+ cd "$MODEL_DIR"
14
+
15
+ python -m pip install --no-deps dist/neohorse_decision-1.0.0-py3-none-any.whl
16
+ python -m pip install 'fastapi==0.141.1' 'uvicorn==0.53.0' 'starlette==1.6.0' 'httpx==0.28.1' 'pillow==12.3.0'
17
+ ```
18
+
19
+ `--no-deps` does not install ML dependencies such as Torch. When installing or upgrading Torch, check whether dependency resolution changes the Triton version. Avoid mixing in Transformers or TorchVision packages from other environments.
20
+
21
+ `backbone/` contains both language and vision parameters and occupies approximately 9.08 GB. The separate decision head occupies approximately 5.25 MB. Actual GPU memory usage also depends on the input and runtime settings. Load the complete directory with the matching runtime.
22
+
23
+ ### Install from Bundled Source
24
+
25
+ With the native ML dependencies above already installed, run this from the root of the downloaded model bundle:
26
+
27
+ ```bash
28
+ python -m pip install --no-deps ./package
29
+ ```
30
+
31
+ `MODEL_DIR` still points to the complete model bundle downloaded from Hugging Face or ModelScope. The source installation replaces the wheel installation step above.
32
+
33
+ ### Model Composition
34
+
35
+ The unified multimodal backbone contains the language model, vision encoder, and merger (`Qwen3_5Model`, with `language_model` and `visual` components). Backbone weights use BF16; the separate pointer head uses FP32. Keep weights, tokenizer, configuration, and runtime from the same release together. Release provenance is recorded in [model_manifest.json](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/model_manifest.json).
36
+
37
+ ## 2. Local Text Inference
38
+
39
+ The CLI can read the example request included in the model release:
40
+
41
+ ```bash
42
+ CUDA_VISIBLE_DEVICES=0 neohorse-decision predict \
43
+ --model-dir "$MODEL_DIR" --request "$MODEL_DIR/example_request.json"
44
+ ```
45
+
46
+ Change the GPU index to match your allocation. The runtime does not schedule GPU resources across a cluster.
47
+
48
+ ### Python Decision Examples
49
+
50
+ **Provide a state and get a yes/no probability, a selected candidate, or a rating.** The examples below use the same user message to demonstrate the three decision modes.
51
+
52
+ Load the model once, then reuse `engine` and `state`:
53
+
54
+ ```python
55
+ import os
56
+
57
+ from neohorse_decision import DecisionEngine
58
+
59
+ engine = DecisionEngine(os.environ["MODEL_DIR"])
60
+ state = "I was charged twice for the same order. Please refund the extra charge today."
61
+ ```
62
+
63
+ All output numbers below are illustrative, not measured results. Actual values depend on the model's predictions.
64
+
65
+ #### Noul: Is It True?
66
+
67
+ **Is the user requesting a refund?** Return the probability of "yes", `P(true)`.
68
+
69
+ ```python
70
+ result = engine.predict({
71
+ "state": state,
72
+ "questions": {
73
+ "refund": {
74
+ "type": "noul",
75
+ "instructions": "Is the user requesting a refund?",
76
+ },
77
+ },
78
+ })
79
+ print(result["answers"]["refund"]["noul"])
80
+ ```
81
+
82
+ Illustrative output: `0.97` means the model assigns a 97% probability to the user requesting a refund. Your application can use this to enter a refund workflow.
83
+
84
+ #### Choice: Which One?
85
+
86
+ **Which team should handle this message?** Select from the candidates and return each candidate's probability.
87
+
88
+ ```python
89
+ result = engine.predict({
90
+ "state": state,
91
+ "questions": {
92
+ "team": {
93
+ "type": "choice",
94
+ "instructions": "Which team should handle this message?",
95
+ "criteria": {
96
+ "billing": "Billing, charges, or refunds",
97
+ "technical": "Product failures or technical issues",
98
+ "other": "Other matters",
99
+ },
100
+ },
101
+ },
102
+ })
103
+ print(result["answers"]["team"]["choice"])
104
+ print(result["answers"]["team"]["probabilities"])
105
+ ```
106
+
107
+ Illustrative output:
108
+
109
+ ```text
110
+ billing
111
+ {'billing': 0.96, 'technical': 0.03, 'other': 0.01}
112
+ ```
113
+
114
+ Read `billing` to route the message to the billing team.
115
+
116
+ #### Score: To What Degree?
117
+
118
+ **How urgent is the request?** Rate it against the levels you define. Levels start at `0`, and the result is their probability-weighted expected value.
119
+
120
+ ```python
121
+ result = engine.predict({
122
+ "state": state,
123
+ "questions": {
124
+ "urgency": {
125
+ "type": "score",
126
+ "instructions": "How soon does the user want this resolved?",
127
+ "criteria": ["Can wait", "This week", "Today"],
128
+ },
129
+ },
130
+ })
131
+ print(result["answers"]["urgency"]["score"])
132
+ ```
133
+
134
+ Illustrative output: `1.9` is close to level `2` ("Today"), which your application can use to raise the request's priority.
135
+
136
+ Save the four Python blocks above, in order, as `quickstart.py`, then run:
137
+
138
+ ```bash
139
+ CUDA_VISIBLE_DEVICES=0 python quickstart.py
140
+ ```
141
+
142
+ To make all three decisions in one text request, place `refund`, `team`, and `urgency` in the same `questions` dictionary. One request returns three answers. Set decision thresholds using data from your own tasks.
143
+
144
+
145
+ ## 3. Start the HTTP Service
146
+
147
+ ```bash
148
+ CUDA_VISIBLE_DEVICES=0 neohorse-decision serve --model-dir "$MODEL_DIR" --port 8080
149
+ ```
150
+
151
+ The service binds to `127.0.0.1` by default. Check readiness from another terminal:
152
+
153
+ ```bash
154
+ curl -sS http://127.0.0.1:8080/health
155
+ ```
156
+
157
+ | Endpoint | Purpose |
158
+ | --- | --- |
159
+ | `POST /v1/decision` | Native decision API |
160
+ | `POST /v1/systemone` | Response structure containing `model`, `answers`, and `usage` |
161
+ | `GET /health` | Readiness status and `input_modalities` |
162
+
163
+ To enable authentication, set `NEOHORSE_API_KEY` in the service environment before starting it. Clients send `Authorization: Bearer <API_KEY>`. For external access, use a TLS gateway and limit concurrency and request body sizes. Only the endpoints and protocol scope described here are supported; `/v1/models` is not provided.
164
+
165
+ ## 4. Text Requests
166
+
167
+ A request contains `model`, `state`, and `questions`. Each question has an application-defined key, a `type`, `instructions`, and `criteria` where required.
168
+
169
+ | Type | `criteria` | Meaning |
170
+ | --- | --- | --- |
171
+ | `noul` | Optional | Determine whether the condition is true |
172
+ | `choice` | Dictionary of candidate keys and descriptions | Select from the candidates |
173
+ | `score` | List of rating descriptions ordered from lowest to highest | Compute the rating distribution and expected value |
174
+
175
+ This request makes three decisions about the same user message:
176
+
177
+ ```bash
178
+ curl -sS http://127.0.0.1:8080/v1/systemone \
179
+ -H 'Content-Type: application/json' \
180
+ -d '{
181
+ "model": "NeoHorse-Jev-4B",
182
+ "state": "I was charged twice for the same order. Please refund the extra charge today.",
183
+ "questions": {
184
+ "refund": {"type": "noul", "instructions": "Is the user requesting a refund?"},
185
+ "team": {
186
+ "type": "choice",
187
+ "instructions": "Which team should handle this message?",
188
+ "criteria": {"billing": "Billing, charges, or refunds", "technical": "Product failures or technical issues", "other": "Other matters"}
189
+ },
190
+ "urgency": {
191
+ "type": "score",
192
+ "instructions": "How soon does the user want this resolved?",
193
+ "criteria": ["Can wait", "This week", "Today"]
194
+ }
195
+ }
196
+ }'
197
+ ```
198
+
199
+ This example assumes a local service without authentication. If authentication is enabled, add `-H "Authorization: Bearer $NEOHORSE_API_KEY"`.
200
+
201
+ Use `NeoHorse-Jev-4B` or `TokenRhythm/NeoHorse-Jev-4B` as the model name. Both HTTP endpoints also accept `neohorse-jev`, `NeoHorse-JEV-4B`, and `TokenRhythm/NeoHorse-JEV-4B`. Responses use the canonical name `NeoHorse-Jev-4B`.
202
+
203
+ ## 5. Responses
204
+
205
+ Each question's result is available at `answers.<question_key>`:
206
+
207
+ | Type | Main fields |
208
+ | --- | --- |
209
+ | Choice | `type`, `choice`, `probabilities`, `confidence` |
210
+ | Noul | `type`, `noul`; the native API also retains yes/no `probabilities` |
211
+ | Score | `type`, `score`, `legend`, `probabilities`, `confidence` |
212
+
213
+ `noul` is the probability that the condition is true. Score levels are indexed from `0`; `score` is their probability-weighted expected value and can be fractional. `legend` maps level indices to their descriptions.
214
+
215
+ | Endpoint | Top-level structure and token counts |
216
+ | --- | --- |
217
+ | `/v1/decision` | `model`, `answers`, `input_tokens`; image requests also return `image_tokens` |
218
+ | `/v1/systemone` | `model`, `answers`, `usage`; token counts are in `usage.input_tokens` and `usage.output_tokens`, with `usage.image_tokens` for image requests |
219
+
220
+ Both HTTP endpoints use the same weights, encoding, and probabilities. The text Python interface, `DecisionEngine.predict`, does not currently return `confidence`.
221
+
222
+ `input_tokens` counts encoded input tokens. The shared state in a text request with multiple questions is counted once, so this is not the total number of tokens processed by the GPU after expanding the questions into separate branches. `output_tokens` is the response JSON's local tokenizer count; it does not indicate autoregressive generation. For image requests, `input_tokens` already includes image tokens and vision start/end markers. Do not add `image_tokens` again. Text responses do not include an `image_tokens` field.
223
+
224
+ ### Understanding confidence
225
+
226
+ `confidence` is a local distribution statistic, not a calibrated probability of correctness:
227
+
228
+ - Choice: `(max(p) - 1/K) / (1 - 1/K)`, where `K` is the number of candidates. A single candidate returns `1`.
229
+ - Score: `1 - sum_i p_i * abs(i - argmax(p)) / (L - 1)`, where `L` is the number of rating levels.
230
+
231
+ Values are clamped to `[0, 1]`. Validate application thresholds on independent data. Both endpoints return `X-NeoHorse-Confidence: local-distribution-statistic-v1`. The System One-style endpoint also returns `X-NeoHorse-Usage: local-tokenizer-not-jev-billing`.
232
+
233
+ ## 6. Image Requests
234
+
235
+ Both HTTP endpoints accept an optional top-level `image` field containing a base64 data URL for a PNG, JPEG, or WebP image. Each image request supports one static image and one Noul, Choice, or Score question. Omit `image` when no image is provided; do not send `null`. External URLs and server file paths are not accepted.
236
+
237
+ ### Python Image Decisions
238
+
239
+ **Provide a page screenshot and a task goal to determine whether the task succeeded, what state the page is in, or how far the task has progressed.** The image and text jointly inform the decision, with Noul, Choice, or Score outputs.
240
+
241
+ This is a standalone image example. Save your screenshot as `screenshot.png`, then save the four Python blocks in this section, in order, as `multimodal_quickstart.py`. Load the model once and reuse the same image for all three calls, with **one question per request**. The output values below are illustrative, not measured results.
242
+
243
+ ```python
244
+ import os
245
+
246
+ from PIL import Image
247
+ from neohorse_decision.vision import VisionDecisionEngine
248
+
249
+ vision_engine = VisionDecisionEngine(os.environ["MODEL_DIR"])
250
+ with Image.open("screenshot.png") as source:
251
+ screenshot = source.convert("RGB")
252
+ state = "Goal: submit the form. Assess the current page screenshot."
253
+ ```
254
+
255
+ #### Noul: Was the Form Submitted Successfully?
256
+
257
+ ```python
258
+ result = vision_engine.predict({
259
+ "model": "NeoHorse-Jev-4B",
260
+ "state": state,
261
+ "questions": {
262
+ "submitted": {
263
+ "type": "noul",
264
+ "instructions": "Does the screenshot clearly show that the form was submitted successfully?",
265
+ },
266
+ },
267
+ }, screenshot)
268
+ print(result["answers"]["submitted"]["noul"])
269
+ ```
270
+
271
+ For example, `0.97` means the model assigns a 97% probability to the screenshot showing a successful submission. Your workflow can use this to decide whether to move to the next step.
272
+
273
+ #### Choice: What State Is the Page In?
274
+
275
+ ```python
276
+ result = vision_engine.predict({
277
+ "model": "NeoHorse-Jev-4B",
278
+ "state": state,
279
+ "questions": {
280
+ "page_status": {
281
+ "type": "choice",
282
+ "instructions": "Which page state does the screenshot show?",
283
+ "criteria": {
284
+ "success": "Submission succeeded",
285
+ "error": "Submission failed or an error is shown",
286
+ "processing": "Submission or loading is in progress",
287
+ "unknown": "Cannot determine the submission status from the screenshot",
288
+ },
289
+ },
290
+ },
291
+ }, screenshot)
292
+ print(result["answers"]["page_status"]["choice"])
293
+ print(result["answers"]["page_status"]["probabilities"])
294
+ ```
295
+
296
+ For example, the result may be `success` alongside each candidate's probability. Route the next step according to the selected state.
297
+
298
+ #### Score: How Far Has the Task Progressed?
299
+
300
+ ```python
301
+ result = vision_engine.predict({
302
+ "model": "NeoHorse-Jev-4B",
303
+ "state": state,
304
+ "questions": {
305
+ "completion": {
306
+ "type": "score",
307
+ "instructions": "How far has the form submission task progressed, based on the screenshot?",
308
+ "criteria": ["Submission has not started", "Submission is in progress", "Submission clearly succeeded"],
309
+ },
310
+ },
311
+ }, screenshot)
312
+ print(result["answers"]["completion"]["score"])
313
+ ```
314
+
315
+ For example, `1.9` is close to level `2` ("Submission clearly succeeded"). Actual results depend on the input image.
316
+
317
+ ```bash
318
+ CUDA_VISIBLE_DEVICES=0 python multimodal_quickstart.py
319
+ ```
320
+
321
+ Image requests currently support **one static image + text + one question**. The HTTP image formats are PNG, JPEG, and WebP. Multiple images, video, and audio are not supported.
322
+
323
+ For local image inference, run:
324
+
325
+ ```bash
326
+ CUDA_VISIBLE_DEVICES=0 python "$MODEL_DIR/vision/example.py" \
327
+ --model-dir "$MODEL_DIR" --image /path/to/image.png \
328
+ --request "$MODEL_DIR/vision/example_request.json"
329
+ ```
330
+
331
+ The Python interface is `neohorse_decision.vision.VisionDecisionEngine`, called as `engine.predict(request, pil_image)`. HTTP text and image requests share the same backbone, decision head, and GPU lock, without dynamic batching. The result is a structured decision distribution. Video and multiple-image interfaces are not provided. For the separate vLLM and SGLang adapters, including text and single-image requests, see the [backend deployment guide](https://github.com/TokenRhythm/NeoHorse/blob/main/jev/infer/README.md).
332
+
333
+ ### HTTP Image Requests
334
+
335
+ For the same question about whether the screenshot shows a successful submission, save this as `screenshot_request.json`:
336
+
337
+ ```json
338
+ {
339
+ "model": "NeoHorse-Jev-4B",
340
+ "state": "Goal: submit the form. Assess the current page screenshot.",
341
+ "questions": {
342
+ "submitted": {
343
+ "type": "noul",
344
+ "instructions": "Does the screenshot clearly show that the form was submitted successfully?"
345
+ }
346
+ }
347
+ }
348
+ ```
349
+
350
+ Once the service is running, use the bundled client to read the local screenshot and send the request:
351
+
352
+ ```bash
353
+ python "$MODEL_DIR/vision/http_example.py" \
354
+ --image screenshot.png \
355
+ --request screenshot_request.json \
356
+ --base-url http://127.0.0.1:8080 \
357
+ --endpoint systemone
358
+ ```
359
+
360
+ The client encodes the image as a base64 data URL in the top-level `image` field. In the response, `answers.submitted.noul` is the probability of a successful submission. Set `--endpoint` to `decision` to use the native endpoint. When authentication is enabled, the client reads the key from the `NEOHORSE_API_KEY` environment variable.
361
+
362
+ If `--request` is omitted, the bundled client asks for the image's dominant color by default.
363
+
364
+ ## 7. Request Limits and Error Handling
365
+
366
+ | Item | Default limit |
367
+ | --- | --- |
368
+ | Text `state` | 2,048 tokens |
369
+ | Each question branch | 8,192 tokens |
370
+ | Questions per text request | 16 |
371
+ | Total tokens after expanding a text request into question branches | 32,768 |
372
+ | Text HTTP request body; JSON fields other than `image` in an image request | 1 MiB |
373
+ | Image HTTP request body | 8 MiB |
374
+ | Image file after base64 decoding | 4 MiB |
375
+ | Image pixel count | 4,194,304 |
376
+ | Minimum/maximum image preprocessing area budget | 65,536 / 1,048,576 pixels |
377
+ | Image tokens | 1,024 |
378
+ | Total encoded length of an image request | 12,288 tokens |
379
+ | Score levels on `/v1/systemone` | 2–10 |
380
+
381
+ Inputs that exceed these limits are rejected without silent truncation. These are deployment protection limits. Check task performance and GPU memory usage before changing them.
382
+
383
+ | HTTP status | Meaning and action |
384
+ | --- | --- |
385
+ | `401` | Authentication failed; check the Bearer token |
386
+ | `413` | The request body or text fields exceed size limits |
387
+ | `422` | Invalid JSON or fields, unknown model, or unsupported image format, dimensions, token count, multiple questions, animation, or other input constraint violations |
388
+ | `429` | The GPU worker for native `/v1/decision` is busy |
389
+ | `529` | The GPU worker for `/v1/systemone` is busy |
390
+
391
+ Busy responses include `Retry-After: 1`. Clients should back off and retry. There is no separate quota-based rate limiter; control high concurrency at the gateway and test it for your deployment.
392
+
393
+ ## 8. Limitations
394
+
395
+ - **Decisions can be wrong.** Valid structure and normalized probabilities do not guarantee correct judgments. Missing evidence, candidate descriptions, candidate order, and domain shifts can all affect results.
396
+ - **Validate probabilities for your application.** NLL, Brier, and ECE calibration results have not been reported. Set thresholds on an independent dataset.
397
+ - **Scope claims to measured evidence.** Comprehensive evaluations of multilingual inputs, long inputs, and computational isolation between questions are not yet available. Multiple questions in one request do not imply a single shared forward pass.
398
+ - **Applications enforce execution constraints.** Tool permissions, business rules, and action validation remain the application's responsibility. The current materials do not provide latency, GPU memory, or cost comparisons under a common timing protocol.
README.md ADDED
@@ -0,0 +1,584 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ pipeline_tag: text-classification
4
+ library_name: pytorch
5
+ base_model:
6
+ - TokenRhythm/NeoHorse-1-4B
7
+ base_model_relation: finetune
8
+ tags:
9
+ - agentic
10
+ - decision-model
11
+ - typed-decisions
12
+ - structured-prediction
13
+ - non-generative
14
+ - multimodal
15
+ - vision-language
16
+ ---
17
+
18
+ <div align="center">
19
+ <h1>NeoHorse-Jev-4B</h1>
20
+ <p><b>Prefill-only decisions for agent workflows.</b></p>
21
+ </div>
22
+
23
+ <div align="center">
24
+ <a href="https://github.com/TokenRhythm/NeoHorse"><img alt="GitHub" src="https://img.shields.io/badge/GitHub-NeoHorse-181717?logo=github&logoColor=white"></a>
25
+ <a href="https://huggingface.co/collections/TokenRhythm/neohorse-jev"><img alt="Hugging Face" src="https://img.shields.io/badge/Hugging%20Face-Models-FFD21E?logo=huggingface&logoColor=000000"></a>
26
+ <a href="https://www.modelscope.cn/models/TokenRhythm/NeoHorse-Jev-4B"><img alt="ModelScope" src="https://img.shields.io/badge/ModelScope-Models-624AFF?logo=modelscope&logoColor=white"></a>
27
+ <a href="https://tokenrhythm.ai/"><img alt="Company" src="https://img.shields.io/badge/Company-TokenRhythm-F97316?logo=homeassistant&logoColor=white"></a>
28
+ <a href="https://x.com/opensquilla"><img alt="Twitter / X" src="https://img.shields.io/badge/Twitter%20%2F%20X-OpenSquilla-111827?logo=x&logoColor=white"></a>
29
+ <a href="https://github.com/TokenRhythm/NeoHorse/blob/main/jev/LICENSE"><img alt="License: Apache-2.0" src="https://img.shields.io/badge/License-Apache--2.0-64748B"></a>
30
+ </div>
31
+
32
+ <div align="center"><a href="https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md">Deployment</a></div>
33
+
34
+ ## Introduction
35
+
36
+ We introduce **NeoHorse-Jev-4B**, a **4B structured decision model** from TokenRhythm, built on [NeoHorse-1-4B](https://huggingface.co/TokenRhythm/NeoHorse-1-4B). Given a state and questions defined by your application, it predicts decisions and their probabilities for routing requests, selecting tools, checking conditions, and rating outcomes.
37
+
38
+ The model uses **prefill-only inference** with three decision types: **Choice**, **Noul**, and **Score**. It predicts directly over the answers you define, without autoregressive text generation.
39
+
40
+ **NeoHorse-Jev-4B scores 77.70 on the six-group text aggregate below, the highest among the four open-weight decision models with complete results in this comparison.** It also achieves **83.26% mean accuracy** across Nimble, VitaminC, and MASSIVE, **11.50 percentage points** above the NeoHorse-1-4B baseline.
41
+
42
+ - **Application-defined decisions.** Define candidate actions, yes/no questions, or ordered rating levels. Text requests can include multiple questions.
43
+ - **Probabilities for application logic.** Use candidate distributions, yes/no probabilities, and expected ratings to drive routing rules and thresholds.
44
+ - **Local deployment.** Run with vLLM, SGLang, or the native Python, CLI, and HTTP runtime. Optional image requests combine a single image with text.
45
+
46
+ ## Decision Demos
47
+
48
+ ![NeoHorse-Jev-4B six-demo grid](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/resolve/main/assets/jev-six-demo-grid.gif)
49
+
50
+ **Six decision demos:** Tetris, Snake, robot manipulation, Mahjong, four-player bomb arena, and autonomous driving (left to right, top to bottom). Each panel preserves the original replay and decision displays and loops independently.
51
+
52
+ ## Evaluation
53
+
54
+ Results updated **September 24, 2026**. These are our evaluations under the protocols described below. Text accuracy, image understanding, and interactive games are reported separately.
55
+
56
+ ### Text Decision Benchmarks
57
+
58
+ All component scores are on a 0–100 scale; higher is better. NeoHorse-Jev-4B uses the **vLLM** results for JevBench, Kev, and OpenJev in this table; Nimble, VitaminC, and MASSIVE retain the original fixed-subset evaluation results.
59
+
60
+ | Model | JevBench | Kev | OpenJev text | Nimble | VitaminC | MASSIVE | AVG |
61
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
62
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **77.13** | 77.87 | **65.39** | <ins>80.50</ins> | 68.28 | 84.86 | <ins>75.67</ins> |
63
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 73.71 | <ins>81.47</ins> | 54.75 | 73.40 | 76.46 | **85.71** | 74.25 |
64
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 55.82 | 61.30 | 40.07 | 45.04 | **78.63** | 68.57 | 58.24 |
65
+ | [Laya Typed Decisions](https://huggingface.co/convaiinnovations/laya-typed-decisions) | -- | -- | -- | 48.94 | <ins>78.30</ins> | 65.43 | -- |
66
+ | **[NeoHorse-1-4B](https://huggingface.co/TokenRhythm/NeoHorse-1-4B)** | -- | -- | -- | 69.15 | 63.27 | 82.86 | -- |
67
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>75.73</ins> | **81.92** | <ins>58.74</ins> | **87.23** | 77.13 | <ins>85.43</ins> | **77.70** |
68
+
69
+ **Bold scores** mark the best result and <ins>underlined scores</ins> the second-best among the listed open-weight entries. `--` means no result is available. NeoHorse-1-4B is a base-model reference; missing groups are not filled with zeros or results from a different checkpoint.
70
+
71
+ **AVG** is our equal-weight mean of the six displayed group scores, calculated before rounding the final aggregate. It is not pooled per-example accuracy or an official combined leaderboard. Only models with all six groups are ranked by this aggregate; images and games do not enter it.
72
+
73
+ NeoHorse-Jev-4B leads the tested open-weight entries on **Kev (81.92)** and **Nimble (87.23)**. Open-Jev-9B scores higher on JevBench and OpenJev's static text tasks; Kev-4B scores slightly higher on MASSIVE, and the Laya checkpoints score higher on VitaminC. The aggregate advantage therefore reflects the balance across tasks, rather than a win on every benchmark.
74
+
75
+ <details>
76
+ <summary>Benchmark scope, sample counts, and aggregation</summary>
77
+
78
+ | Benchmark group | Evaluated scope | Score used in the overview |
79
+ | --- | --- | --- |
80
+ | JevBench | Public set of 231 examples | Official family-macro score |
81
+ | Kev | Development and test splits of decision-v7, transfer-v4, and transfer-v9; 6,436 input records in total | Equal-weight mean of the six clean-accuracy scores; a record may contain multiple decisions |
82
+ | OpenJev text | Static text tasks: NLI, multiple-choice reranking, and fixed-candidate GSM8K | Equal-weight mean of 19 task scores; the two MNLI splits are averaged first |
83
+ | Nimble | 282 examples selected from 324, keeping related case groups intact; 116 Choice, 112 Noul, 54 Score | Per-example exact decision accuracy, including exact rating-level matches |
84
+ | VitaminC-dev | 599 examples from the upstream Nimble sampling pipeline | Three-way evidence/claim classification accuracy |
85
+ | MASSIVE-en | 350 English test examples from the upstream Nimble sampling pipeline | Classification accuracy across 18 assistant scenarios; not intent/slot or multilingual evaluation |
86
+
87
+ For Nimble, VitaminC, and MASSIVE, selected IDs and records were frozen before model comparison. Reference answers are used for scoring, not as model input. Upstream VitaminC/MASSIVE sampling uses seed `20260918` and complete case groups. Local selection uses a 384-token state limit, a 2,048-token packed decision limit, and at most 26 candidates; the base-model letter-logit prompt allows 4,096 tokens. These are subset-selection rules for these three datasets, not universal limits for all benchmarks or deployment. Nimble falls from 324 to 282 examples after length and whole-group filtering; the other two subsets pass unchanged.
88
+
89
+ </details>
90
+
91
+ <details>
92
+ <summary>Nimble, VitaminC, and MASSIVE: three-benchmark means</summary>
93
+
94
+ | Model | Three-benchmark mean accuracy (%) |
95
+ | --- | ---: |
96
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 77.88 |
97
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | <ins>78.53</ins> |
98
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 64.08 |
99
+ | [Laya Typed Decisions](https://huggingface.co/convaiinnovations/laya-typed-decisions) | 64.22 |
100
+ | **[NeoHorse-1-4B](https://huggingface.co/TokenRhythm/NeoHorse-1-4B)** | 71.76 |
101
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | **83.26** |
102
+
103
+ This mean weights Nimble, VitaminC, and MASSIVE equally, rather than pooling their 1,231 examples. The three-benchmark means retain the original evaluation report, which averages unrounded accuracies; recomputing from the two-decimal component scores can differ by 0.01. For example, Kev is reported as 78.53. The separately defined AVG above uses the six displayed group scores.
104
+
105
+ </details>
106
+
107
+ <details>
108
+ <summary>Detailed text comparisons: JevBench, Kev, and OpenJev</summary>
109
+
110
+ **JevBench.** NeoHorse-Jev-4B has 75.32% per-example accuracy and 100% valid output format on the public 231 examples. Its 75.73 family-macro score weights families, rather than individual examples.
111
+
112
+ | Model | adequacy | adversarial | ambiguous | extraction | fact | intent |
113
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
114
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **83.33** | **100.00** | 42.86 | 91.67 | **100.00** | **100.00** |
115
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 66.67 | **100.00** | <ins>57.14</ins> | <ins>95.83</ins> | **100.00** | <ins>95.83</ins> |
116
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 66.67 | <ins>50.00</ins> | 14.29 | 83.33 | <ins>83.33</ins> | 83.33 |
117
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>75.00</ins> | **100.00** | **71.43** | **100.00** | **100.00** | **100.00** |
118
+
119
+ | Model | judge_hard | long_policy | multi_hop | ordinal | policy | probability |
120
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
121
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **76.47** | **47.37** | **66.67** | **100.00** | **100.00** | **60.00** |
122
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | <ins>52.94</ins> | <ins>21.05</ins> | <ins>55.56</ins> | **100.00** | <ins>91.67</ins> | <ins>50.00</ins> |
123
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 41.18 | <ins>21.05</ins> | 33.33 | <ins>91.67</ins> | 83.33 | <ins>50.00</ins> |
124
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>52.94</ins> | **47.37** | <ins>55.56</ins> | **100.00** | **100.00** | 40.00 |
125
+
126
+ | Model | routing | routing_hard | temporal_numeric | tool_selection | tradeoff | trap |
127
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
128
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | <ins>66.67</ins> | **100.00** | <ins>20.00</ins> | **100.00** | <ins>33.33</ins> | **100.00** |
129
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | **100.00** | **100.00** | 6.67 | **100.00** | <ins>33.33</ins> | **100.00** |
130
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 50.00 | <ins>20.00</ins> | **33.33** | **100.00** | **100.00** | 0.00 |
131
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | **100.00** | **100.00** | 0.00 | **100.00** | <ins>33.33</ins> | <ins>87.50</ins> |
132
+
133
+ Examples: adequacy: 12; adversarial: 6; ambiguous: 7; extraction: 24; fact: 12; intent: 24; judge_hard: 17; long_policy: 19; multi_hop: 18; ordinal: 12; policy: 12; probability: 10; routing: 12; routing_hard: 5; temporal_numeric: 15; tool_selection: 12; tradeoff: 6; trap: 8.
134
+
135
+ **Kev.** Clean accuracy (%) by suite; record counts differ from decision counts.
136
+
137
+ | Model | decision-v7 / development | decision-v7 / test | transfer-v4 / development | transfer-v4 / test | transfer-v9 / development | transfer-v9 / test |
138
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
139
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 80.30 | 77.00 | 77.44 | 83.54 | 72.47 | <ins>76.48</ins> |
140
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | **87.18** | **87.08** | <ins>79.73</ins> | <ins>83.69</ins> | <ins>74.76</ins> | 76.39 |
141
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 66.22 | 65.50 | 65.09 | 65.55 | 52.39 | 53.06 |
142
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>86.23</ins> | <ins>86.58</ins> | **81.71** | **84.60** | **75.53** | **76.86** |
143
+
144
+ Input records: decision-v7 / development: 1,204; decision-v7 / test: 1,176; transfer-v4 / development: 764; transfer-v4 / test: 764; transfer-v9 / development: 1,264; transfer-v9 / test: 1,264.
145
+
146
+ **OpenJev static text.** NLI classification accuracy (%):
147
+
148
+ | Model | scitail | anli_r1 | anli_r2 | anli_r3 | wanli | control |
149
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
150
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 79.16 | **74.00** | **66.30** | **59.42** | **67.10** | **67.58** |
151
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | <ins>84.81</ins> | 65.60 | 54.30 | 52.25 | 63.50 | 64.35 |
152
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 74.84 | 48.40 | 38.80 | 34.33 | 53.00 | 37.64 |
153
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | **87.02** | <ins>67.20</ins> | <ins>56.20</ins> | <ins>53.42</ins> | <ins>65.74</ins> | <ins>65.96</ins> |
154
+
155
+ | Model | MNLI / validation_matched | MNLI / validation_mismatched |
156
+ | --- | ---: | ---: |
157
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 80.64 | 80.54 |
158
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | **89.17** | <ins>89.35</ins> |
159
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 63.28 | 64.27 |
160
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>88.95</ins> | **89.46** |
161
+
162
+ Examples: scitail: 2,126; anli_r1: 1,000; anli_r2: 1,000; anli_r3: 1,200; wanli: 5,000; control: 805; MNLI / validation_matched: 9,815; MNLI / validation_mismatched: 9,832.
163
+
164
+ Multiple-choice rerank accuracy:
165
+
166
+ | Model | arc_easy | arc_challenge | winogrande | gsm8k_mc4 | gsm8k_mc10 | gpqa |
167
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
168
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **95.71** | **87.29** | **66.30** | **52.69** | **35.71** | **37.37** |
169
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 75.42 | 66.89 | 58.33 | 38.59 | 20.77 | 33.84 |
170
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 40.19 | 33.36 | 49.64 | 24.26 | 8.49 | 26.77 |
171
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>88.93</ins> | <ins>78.50</ins> | <ins>60.69</ins> | <ins>41.77</ins> | <ins>21.83</ins> | <ins>34.34</ins> |
172
+
173
+ | Model | gpqa_fewshot | chess | hellaswag | mmlu | mmlu_fewshot |
174
+ | --- | ---: | ---: | ---: | ---: | ---: |
175
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **39.90** | **52.40** | **54.09** | **66.80** | **65.05** |
176
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | <ins>34.85</ins> | 19.20 | 18.92 | 52.29 | 57.50 |
177
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 22.22 | <ins>29.60</ins> | 28.44 | 29.96 | 26.58 |
178
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | 34.34 | 22.00 | <ins>34.95</ins> | <ins>59.19</ins> | <ins>60.20</ins> |
179
+
180
+ Examples: arc_easy: 2,376; arc_challenge: 1,172; winogrande: 1,267; gsm8k_mc4: 1,319; gsm8k_mc10: 1,319; gpqa: 198; gpqa_fewshot: 198; chess: 500; hellaswag: 10,042; mmlu: 14,042; mmlu_fewshot: 14,042.
181
+
182
+ **GSM8K with frozen candidates (200 examples).** The main metric is `nli_rerank@4`:
183
+
184
+ | Model | nli_rerank@4 | nli_rerank_margin@4 |
185
+ | --- | ---: | ---: |
186
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | **95.00** | **95.00** |
187
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 89.50 | 89.50 |
188
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 91.00 | 91.50 |
189
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | <ins>94.50</ins> | <ins>94.00</ins> |
190
+
191
+ Examples: nli_rerank@4: 200; nli_rerank_margin@4: 200.
192
+
193
+ The shared candidate set has 93.00% greedy accuracy, 93.50% majority-vote accuracy, and a 97.00% oracle@4 ceiling. These are properties of the same candidate pool, not separate generations by each decision model. The 19-task aggregate uses the main rerank score, not the auxiliary margin score.
194
+
195
+ </details>
196
+
197
+ ### Image and Text Evaluation
198
+
199
+ On **Image-NLI**, NeoHorse-Jev-4B reaches **60.65% accuracy over 8,000 examples** with vLLM; the native runtime gives 60.66%. The task evaluates statements against an image and text context.
200
+
201
+ | Model | Overall | `vqa_answer` | `vqa_answer_neg` | `vqa_disagree` | `vqa_spatial` | `vqa_yesno` |
202
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
203
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>vLLM | 60.65 | 75.94 | 68.79 | 52.88 | 59.95 | 48.53 |
204
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>Native | 60.66 | 75.94 | 68.63 | 52.98 | 60.03 | 48.37 |
205
+
206
+ Examples: Overall: 8,000; `vqa_answer`: 1376; `vqa_answer_neg`: 644; `vqa_disagree`: 1040; `vqa_spatial`: 3648; `vqa_yesno`: 1292.
207
+
208
+ The comparison report contains no Image-NLI results for Kev-4B, Open-Jev-9B, or Laya, so this is a capability measurement without a cross-model ranking. The image assets were reconstructed and frozen locally; this does not claim reproduction of the upstream author's unavailable original image assets.
209
+
210
+ <details>
211
+ <summary>Doom with image input: all 11 candidate configurations</summary>
212
+
213
+ | Model | `action` | `danger` | `pixels` | `pixels_pct` | `pixels_sym` | `precise` |
214
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
215
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>vLLM | 1.00 | 1.40 | 9.20 | 7.40 | 8.40 | 16.00 |
216
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>Native | 1.00 | 1.40 | 11.80 | 8.80 | 10.00 | 15.60 |
217
+
218
+ | Model | `should` | `thirds` | `where` | `where_closest` | `where_plain` |
219
+ | --- | ---: | ---: | ---: | ---: | ---: |
220
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>vLLM | 1.00 | 12.80 | 1.40 | 1.40 | 6.00 |
221
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>Native | 1.00 | 10.60 | 1.40 | 1.40 | 8.40 |
222
+
223
+ Each configuration runs for five episodes; scores are mean kills. Five author configurations (`pixels`, `pixels_sym`, `precise`, `thirds`, `where_closest`) use the completed reruns; the six unchanged configurations retain their valid results. The report supplies no image-interface results for the comparison models. All configurations are listed because candidate wording substantially affects the outcome.
224
+
225
+ </details>
226
+
227
+ ### Interactive Decision Tasks
228
+
229
+ The September 24 results include the completed game reruns and corrected Minecraft action execution. The tables below report the specified candidate configurations separately and use vLLM for NeoHorse-Jev-4B unless another backend is named. **Text-state Doom, Flappy, and Minecraft results are not image-input evaluations.** Game scores use their own units and are excluded from the text aggregate.
230
+
231
+ <details>
232
+ <summary>Cross-model game results: Doom, Flappy, and Minecraft</summary>
233
+
234
+ **Doom with text state — mean kills, five episodes per configuration.**
235
+
236
+ | Model | `position` (author configuration) | `position_none` (includes no-enemy condition) | `aligned_state` (aligned target and tolerance) |
237
+ | --- | ---: | ---: | ---: |
238
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 11.20 | 7.80 | 18.60 |
239
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 1.40 | 10.40 | 14.20 |
240
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 1.00 | 1.60 | 1.00 |
241
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | 1.40 | 10.60 | 14.40 |
242
+
243
+ `aligned_state` changes target definition and tolerance wording, so it is a different decision policy from the author's `position` configuration. The reported environment controls are random: 1.00 and oracle: 16.60 mean kills; five-episode outcomes should not be read as a precise ranking.
244
+
245
+ **Flappy — mean pipes cleared.** Both NeoHorse-Jev backends are shown because real-time outcomes depend on the deployment path. No best/second-best markers are applied to this timing-dependent table.
246
+
247
+ | Model | sign (author) | position (author) | action (real-time) | action (wait for model) | position_v (real-time) | position_v (wait for model) |
248
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
249
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 28.00 | 23.50 | 0.60 | 0.70 | 8.95 | 48.00 |
250
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 28.00 | 2.67 | 0.00 | 0.10 | 26.80 | 48.00 |
251
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 19.00 | 0.50 | 0.00 | 0.00 | 0.00 | 0.00 |
252
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>vLLM | 27.67 | 27.67 | 0.05 | 0.05 | 15.35 | 48.00 |
253
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)**<br>Native | 28.00 | 27.83 | 0.00 | 0.05 | 42.00 | 48.00 |
254
+
255
+ Settings: sign (author) and position (author) — 6 episodes, 900-frame cap, 15 FPS; action (real-time) and position_v (real-time) — 20 episodes, 1500-frame cap, 30 FPS; action (wait for model) and position_v (wait for model) — 20 episodes, 1500-frame cap, 0 FPS.
256
+
257
+ `FPS = 0` waits for every model's response. Flappy outcomes are not a controlled cross-model speed benchmark.
258
+
259
+ **Real Minecraft — success rate (%), ten episodes and at most 60 decision steps per strategy.**
260
+
261
+ | Model | flat/action | flat/state | chain |
262
+ | --- | ---: | ---: | ---: |
263
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 0.00 | 80.00 | 100.00 |
264
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 0.00 | 60.00 | 90.00 |
265
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 0.00 | 0.00 | 20.00 |
266
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | 20.00 | 50.00 | 90.00 |
267
+
268
+ These use the corrected action cancellation and pathfinding-failure handling, with a 240-second action timeout. `flat/action` is a separate ten-episode run; `flat/state` and `chain` share the corrected execution setup. Native NeoHorse-Jev results are 30.00%, 40.00%, and 90.00%, respectively. The environment oracle itself reaches 70–100% across model runs, while random scores 0%, so environment variation remains relevant.
269
+
270
+ **Simulated Minecraft — success rate (%), ten episodes and at most 60 decision steps per strategy.**
271
+
272
+ | Model | flat/action | chain |
273
+ | --- | ---: | ---: |
274
+ | [Open-Jev-9B](https://huggingface.co/ZefanCai/Open-Jev-9B) | 0.00 | 100.00 |
275
+ | [Kev-4B](https://huggingface.co/jaredpalmer/kev-4b) | 0.00 | 100.00 |
276
+ | [Laya English](https://huggingface.co/convaiinnovations/laya) | 0.00 | 0.00 |
277
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | 0.00 | 100.00 |
278
+
279
+ The simulated environment's oracle reaches 100% and random scores 0%. Simulated and real Minecraft are different settings and should not be averaged together.
280
+
281
+ </details>
282
+
283
+ ## Download Model
284
+
285
+ | Model | Download Links | Parameters | Base Model |
286
+ | --- | --- | --- | --- |
287
+ | **[NeoHorse-Jev-4B](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)** | [🤗 Hugging Face](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B)<br>[🤖 ModelScope](https://www.modelscope.cn/models/TokenRhythm/NeoHorse-Jev-4B) | ~4B | [NeoHorse-1-4B](https://huggingface.co/TokenRhythm/NeoHorse-1-4B) |
288
+
289
+ Download the complete model bundle, including the backbone, tokenizer, separate decision head, and matching runtime wheel. The [GitHub source repository](https://github.com/TokenRhythm/NeoHorse/tree/main/jev) provides the inference source, backend adapters, and examples.
290
+
291
+ <details>
292
+ <summary>Model details</summary>
293
+
294
+ | Field | Value |
295
+ | --- | --- |
296
+ | Parameters | Approximately 4B |
297
+ | Base model | NeoHorse-1-4B |
298
+ | Input | Text, or a single image with text |
299
+ | Decision types | Choice, Noul, Score |
300
+ | Inference | Prefill-only |
301
+ | License | Apache-2.0 |
302
+
303
+ </details>
304
+
305
+ ## Deployment
306
+
307
+ Choose [vLLM](#vllm), [SGLang](#sglang), or the [native runtime](#native-runtime). Each path requires the complete model bundle from Hugging Face or ModelScope.
308
+
309
+ Use a separate, existing environment for each backend. The adapters and example requests are maintained in the [GitHub source repository](https://github.com/TokenRhythm/NeoHorse/tree/main/jev). If you have only downloaded the model bundle, obtain the source first:
310
+
311
+ ```bash
312
+ git clone https://github.com/TokenRhythm/NeoHorse.git
313
+ cd NeoHorse/jev
314
+ ```
315
+
316
+ Run the commands below from the `jev/` directory of the cloned NeoHorse repository and replace `/path/to/model` with the complete model directory downloaded from Hugging Face or ModelScope.
317
+
318
+ ### vLLM
319
+
320
+ Use an existing **vLLM 0.28.0** environment.
321
+
322
+ ```bash
323
+ # Start the server and keep this terminal running
324
+ CUDA_VISIBLE_DEVICES=0 python infer/vllm/launch.py \
325
+ --bundle /path/to/model --port 30000
326
+
327
+ # Once ready, run inference from another terminal
328
+ python infer/vllm/infer.py \
329
+ --bundle /path/to/model \
330
+ --url http://127.0.0.1:30000 \
331
+ --request infer/request.json
332
+ ```
333
+
334
+ ### SGLang
335
+
336
+ Use an existing **SGLang 0.5.17** environment.
337
+
338
+ ```bash
339
+ # Start the server and keep this terminal running
340
+ CUDA_VISIBLE_DEVICES=0 python infer/sglang/launch.py \
341
+ --bundle /path/to/model --port 30000
342
+
343
+ # Once ready, run inference from another terminal
344
+ python infer/sglang/infer.py \
345
+ --bundle /path/to/model \
346
+ --url http://127.0.0.1:30000 \
347
+ --request infer/request.json
348
+ ```
349
+
350
+ The sample request is included. Results are printed to the terminal; read `answers.move.choice` and `answers.move.probabilities`. Both backends also support a single image combined with text. See [infer/README.md](https://github.com/TokenRhythm/NeoHorse/blob/main/jev/infer/README.md) for text and image examples, dependency setup, and input limits.
351
+
352
+ ### Native Runtime
353
+
354
+ The `neohorse_decision` package provides local Python and CLI inference, plus HTTP services for text and image decisions. Expand the walkthrough for installation and examples of all three decision types, or see the [Deployment](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md) guide for the complete API reference.
355
+
356
+ <details>
357
+ <summary>Installation and usage examples</summary>
358
+
359
+ #### 1. Download the Complete Model Release
360
+
361
+ Download the complete release from [Hugging Face](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/tree/main) or [ModelScope](https://www.modelscope.cn/models/TokenRhythm/NeoHorse-Jev-4B), then set its local path:
362
+
363
+ ```bash
364
+ export MODEL_DIR="/path/to/NeoHorse-Jev-4B"
365
+ cd "$MODEL_DIR"
366
+ ```
367
+
368
+ The complete model bundle contains:
369
+
370
+ | File or directory | Purpose |
371
+ | --- | --- |
372
+ | `backbone/` | Unified multimodal backbone; language and vision parameters share safetensors shards and an index |
373
+ | `tokenizer/` | Matching tokenizer |
374
+ | `pointer_head.safetensors` | Separate decision head |
375
+ | `model_manifest.json` | Model composition and provenance |
376
+ | `dist/`, `package/` | Runtime wheel and source |
377
+ | `example_request.json` | Example request covering all three decision types |
378
+ | `vision/` | Local image inference and HTTP image client examples |
379
+
380
+ Use the matching `neohorse_decision` package for the native runtime. The vLLM and SGLang adapters are in the [GitHub source repository](https://github.com/TokenRhythm/NeoHorse/tree/main/jev/infer); see [backend deployment](#vllm) above. All three paths require the complete model directory, including the separate decision head.
381
+
382
+ #### 2. Install the Runtime
383
+
384
+ The recorded test environment is **Linux, Python 3.12, PyTorch 2.8.0, Transformers 5.17.0, Triton 3.7.1, and flash-linear-attention 0.5.2**, with a CUDA GPU that supports BF16. See [environment.json](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/environment.json) and [DEPLOYMENT.md](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md) for details.
385
+
386
+ The following commands assume these ML dependencies are already installed in an isolated environment and GPU 0 has been allocated to your workload:
387
+
388
+ ```bash
389
+ python -m pip install --no-deps dist/neohorse_decision-1.0.0-py3-none-any.whl
390
+ python -m pip install 'fastapi==0.141.1' 'uvicorn==0.53.0' 'starlette==1.6.0' 'httpx==0.28.1' 'pillow==12.3.0'
391
+
392
+ CUDA_VISIBLE_DEVICES=0 neohorse-decision predict --model-dir . --request example_request.json
393
+ ```
394
+
395
+ `--no-deps` installs the bundled wheel into an already prepared environment; it does not install the ML dependencies listed above. The unified backbone weights occupy approximately 9.08 GB. Actual GPU memory usage also depends on input and runtime settings. **Download the complete model repository and install its bundled runtime.**
396
+
397
+ #### Decision Types
398
+
399
+ | Type | Input | Output | Typical use cases |
400
+ | --- | --- | --- | --- |
401
+ | **Choice** | An ordered dictionary of candidate keys and descriptions | Selected candidate and full candidate probability distribution | Request routing, tool selection, action selection |
402
+ | **Noul** | A yes/no question | Probability that the statement is true, `P(true)` | Condition checks, filtering, workflow gates |
403
+ | **Score** | Rating levels ordered from lowest to highest | Probability distribution over levels and the expected rating | Quality assessment, severity, priority |
404
+
405
+ Score levels are indexed from `0`, and the expected rating can be fractional. These use cases describe the interface; performance should be validated on your target tasks.
406
+
407
+ #### 3. Python Examples
408
+
409
+ **Provide a state and get a yes/no probability, a selected candidate, or a rating.** The examples below use the same user message to demonstrate the three decision modes.
410
+
411
+ Load the model once, then reuse `engine` and `state`:
412
+
413
+ ```python
414
+ import os
415
+
416
+ from neohorse_decision import DecisionEngine
417
+
418
+ engine = DecisionEngine(os.environ["MODEL_DIR"])
419
+ state = "I was charged twice for the same order. Please refund the extra charge today."
420
+ ```
421
+
422
+ All output numbers below are illustrative, not measured results. Actual values depend on the model's predictions.
423
+
424
+ ##### Noul: Is It True?
425
+
426
+ **Is the user requesting a refund?** Return the probability of "yes", `P(true)`.
427
+
428
+ ```python
429
+ result = engine.predict({
430
+ "state": state,
431
+ "questions": {
432
+ "refund": {
433
+ "type": "noul",
434
+ "instructions": "Is the user requesting a refund?",
435
+ },
436
+ },
437
+ })
438
+ print(result["answers"]["refund"]["noul"])
439
+ ```
440
+
441
+ Illustrative output: `0.97` means the model assigns a 97% probability to the user requesting a refund. Your application can use this to enter a refund workflow.
442
+
443
+ ##### Choice: Which One?
444
+
445
+ **Which team should handle this message?** Select from the candidates and return each candidate's probability.
446
+
447
+ ```python
448
+ result = engine.predict({
449
+ "state": state,
450
+ "questions": {
451
+ "team": {
452
+ "type": "choice",
453
+ "instructions": "Which team should handle this message?",
454
+ "criteria": {
455
+ "billing": "Billing, charges, or refunds",
456
+ "technical": "Product failures or technical issues",
457
+ "other": "Other matters",
458
+ },
459
+ },
460
+ },
461
+ })
462
+ print(result["answers"]["team"]["choice"])
463
+ print(result["answers"]["team"]["probabilities"])
464
+ ```
465
+
466
+ Illustrative output:
467
+
468
+ ```text
469
+ billing
470
+ {'billing': 0.96, 'technical': 0.03, 'other': 0.01}
471
+ ```
472
+
473
+ Read `billing` to route the message to the billing team.
474
+
475
+ ##### Score: To What Degree?
476
+
477
+ **How urgent is the request?** Rate it against the levels you define. Levels start at `0`, and the result is their probability-weighted expected value.
478
+
479
+ ```python
480
+ result = engine.predict({
481
+ "state": state,
482
+ "questions": {
483
+ "urgency": {
484
+ "type": "score",
485
+ "instructions": "How soon does the user want this resolved?",
486
+ "criteria": ["Can wait", "This week", "Today"],
487
+ },
488
+ },
489
+ })
490
+ print(result["answers"]["urgency"]["score"])
491
+ ```
492
+
493
+ Illustrative output: `1.9` is close to level `2` ("Today"), which your application can use to raise the request's priority.
494
+
495
+ Save the four Python blocks above, in order, as `quickstart.py`, then run:
496
+
497
+ ```bash
498
+ CUDA_VISIBLE_DEVICES=0 python quickstart.py
499
+ ```
500
+
501
+ To make all three decisions in one text request, place `refund`, `team`, and `urgency` in the same `questions` dictionary. One request returns three answers. Set decision thresholds using data from your own tasks.
502
+
503
+ #### 4. Image and Text Decisions
504
+
505
+ Save a page screenshot as `screenshot.png`. This Choice example identifies the page's current state:
506
+
507
+ ```python
508
+ import os
509
+
510
+ from PIL import Image
511
+ from neohorse_decision.vision import VisionDecisionEngine
512
+
513
+ vision_engine = VisionDecisionEngine(os.environ["MODEL_DIR"])
514
+ with Image.open("screenshot.png") as source:
515
+ screenshot = source.convert("RGB")
516
+
517
+ result = vision_engine.predict({
518
+ "model": "NeoHorse-Jev-4B",
519
+ "state": "Goal: submit the form. Assess the current page screenshot.",
520
+ "questions": {
521
+ "page_status": {
522
+ "type": "choice",
523
+ "instructions": "Which page state does the screenshot show?",
524
+ "criteria": {
525
+ "success": "Submission succeeded",
526
+ "error": "Submission failed or an error is shown",
527
+ "processing": "Submission or loading is in progress",
528
+ "unknown": "Cannot determine the submission status from the screenshot",
529
+ },
530
+ },
531
+ },
532
+ }, screenshot)
533
+ print(result["answers"]["page_status"]["choice"])
534
+ print(result["answers"]["page_status"]["probabilities"])
535
+ ```
536
+
537
+ Image requests also support Noul and Score, with one image and one question per request. Examples for all three modes and HTTP image requests are in the [image usage guide](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md#6-image-requests).
538
+
539
+ #### 5. HTTP Service
540
+
541
+ Start the service:
542
+
543
+ ```bash
544
+ CUDA_VISIBLE_DEVICES=0 neohorse-decision serve --model-dir "$MODEL_DIR" --port 8080
545
+ ```
546
+
547
+ From another terminal, send the same Noul question:
548
+
549
+ ```bash
550
+ curl -sS http://127.0.0.1:8080/v1/systemone \
551
+ -H 'Content-Type: application/json' \
552
+ -d '{"model":"NeoHorse-Jev-4B","state":"I was charged twice for the same order. Please refund the extra charge today.","questions":{"refund":{"type":"noul","instructions":"Is the user requesting a refund?"}}}'
553
+ ```
554
+
555
+ The service binds to `127.0.0.1` by default. For external access, enable Bearer authentication with `NEOHORSE_API_KEY` and use a TLS gateway. The native endpoint is `/v1/decision`, the System One-style endpoint is `/v1/systemone`, and `/health` reports readiness.
556
+
557
+ See the [deployment and API guide](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md) for request formats, response fields, default limits, and error handling.
558
+
559
+ #### Install from Source
560
+
561
+ With the ML dependencies above already installed, run this from the root of the downloaded model bundle:
562
+
563
+ ```bash
564
+ python -m pip install --no-deps ./package
565
+ ```
566
+
567
+ `MODEL_DIR` still points to the complete model bundle downloaded from Hugging Face or ModelScope. Inference source is in `package/src/neohorse_decision/`; image clients and local image examples are in `vision/`.
568
+
569
+ </details>
570
+
571
+ ## Limitations
572
+
573
+ - **Decisions can be wrong.** Valid structure and normalized probabilities do not guarantee correct judgments. Missing evidence, candidate descriptions, candidate order, and domain shifts can all affect results.
574
+ - **Validate probabilities for your application.** NLL, Brier, and ECE calibration results have not been reported. Set thresholds on an independent dataset.
575
+ - **Scope claims to measured evidence.** Comprehensive evaluations of multilingual inputs, long inputs, and computational isolation between questions are not yet available. Multiple questions in one request do not imply a single shared forward pass.
576
+ - **Applications enforce execution constraints.** Tool permissions, business rules, and action validation remain the application's responsibility. The current materials do not provide latency, GPU memory, or cost comparisons under a common timing protocol.
577
+
578
+ ## License and Acknowledgments
579
+
580
+ NeoHorse-Jev-4B is released under **Apache License 2.0**. It is derived from NeoHorse-1-4B, whose upstream base is Qwen3.5-4B. Bundled third-party runtime components retain their licenses and attribution. Preserve the relevant copyright, license, and modification notices when redistributing.
581
+
582
+ We thank Jared Palmer for open-sourcing [Kev](https://github.com/jaredpalmer/kev). Parts of this project's decision inference code are adapted from Kev.
583
+
584
+ For questions or bug reports, use the [NeoHorse issue tracker](https://github.com/TokenRhythm/NeoHorse/issues).
SHA256SUMS ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 76fdab375b201f5f42d82a06794bdb015d9e98bbe083ada46740b67a26e443bd DEPLOYMENT.md
2
+ 30e9fa2b5abc577cddfc7aa17821226561e53ca22807444d55561273dbf888a0 README.md
3
+ e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855 assets/.gitkeep
4
+ 9d0b6dfa93abc37c50872b9e568b1c06c73ca867bb4592f92429691cca83bebc assets/jev-six-demo-grid.gif
5
+ 3e5547987b4976248251215d90ae234772988a2d3099bdc77c48dc70ba117b0b assets/jev-snake-demo.gif
6
+ 833ef256e6b87f27d0f10cbb06ef51e34b2687922ac1bfabc3bc016166d847ff backbone/config.json
7
+ 7d1adbb748ff60a91b3b6ffba1ff70bfcab855ff2b8cf5e33f9c6bc1ff13cb7e backbone/model-00001-of-00003.safetensors
8
+ c37c278c3977b16b7358421a10b5805615a0f393f19f56b540644dded4d0e6c3 backbone/model-00002-of-00003.safetensors
9
+ 475b9a4b012cc888da1d6575746d2b329a944a6107f8e20b1dcb83b116bf1b1d backbone/model-00003-of-00003.safetensors
10
+ 5c881b4f2ae4b9600d2a7d68a84d4a0dc3c62e6974c601dd281c9e61a3c3da7c backbone/model.safetensors.index.json
11
+ 27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516 backbone/preprocessor_config.json
12
+ 8a8254e1f597af2d400e9df2758d9a6491c3351a92e2c6e70e46eeca780a1e7f config.json
13
+ 7401f1f3a041bd9c05662aaf38ad894e68e0103685a2fd3ed23169c3aff64bfc dist/neohorse_decision-1.0.0-py3-none-any.whl
14
+ 4d5790e2ace3cf8bfc26fbc23c973e2ef1c796911c407d5a75adc48ed4b1c32c environment.json
15
+ 86095b334c58f2c0817e46c53d241ac6fb2503b386e2cb90174480626ddf360b example_request.json
16
+ 8a8254e1f597af2d400e9df2758d9a6491c3351a92e2c6e70e46eeca780a1e7f model_manifest.json
17
+ d6feb31981fc445e3e9e5a18e1ece4909bdec9f4ef293d57dd547e93dcd0127f package/pyproject.toml
18
+ 257213078c9c179fbde33ea8defd63174cd36207bb72a725ea7845b581c30cba package/src/neohorse_decision/__init__.py
19
+ d8283d38f51f2cd88017039dc944d3d44a60e56e871090e077cd23d7e1c11804 package/src/neohorse_decision/_inference.py
20
+ f93060bd0f1875daaee413754eeb3a4757205bdb9e4e7a36b4efc379a027d43f package/src/neohorse_decision/_vendor/LICENSE
21
+ 9bdb2a943d547c457a5614c5ab138b19753798e59f94ed92f3d1c07384510703 package/src/neohorse_decision/_vendor/NOTICE.md
22
+ 7b86e4fe30867d00f45c09e3a9930e7406838c934abd338fb027126aa0eb1f89 package/src/neohorse_decision/_vendor/__init__.py
23
+ cff5be248612012ae49d7e3fd4c75c220c94d5432114a661175a68cf3237cdc1 package/src/neohorse_decision/_vendor/model.py
24
+ e78ee8f180660e2ddb057b208b581f9e1cdca024ea50d87c631286c187691370 package/src/neohorse_decision/_vendor/schema.py
25
+ a3daeb60ce7211e067737fd7345e7b2d20a04403a0424d310a9c38eb0096b1cc package/src/neohorse_decision/cli.py
26
+ 3b6b3f36f830c654ffa161ef222a39b9e766039cfa6d7612add9a4c70fb88b76 package/src/neohorse_decision/engine.py
27
+ 69c64508c024b1217f732c078e74ca7d838e95614d428359698a2f053fddfd41 package/src/neohorse_decision/image_input.py
28
+ bd49282fd749d4550bf83190abdda9fdeb906a86ae2addc9da2c758d92a91286 package/src/neohorse_decision/server.py
29
+ bce2de084b017bb2e5615ea27ffc226e1373cc6410df553a556b88c200c7633c package/src/neohorse_decision/systemone.py
30
+ 3777920631dd0a62d79a4428c06089a898676e45b9468f2dbf430cf109f8e506 package/src/neohorse_decision/vision.py
31
+ 467ae48977b5c4bf87dd1db29021199e0c3401c52c5b57a1eccde679c0bade26 pointer_head.safetensors
32
+ a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715 tokenizer/chat_template.jinja
33
+ 06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523 tokenizer/tokenizer.json
34
+ bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87 tokenizer/tokenizer_config.json
35
+ 50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 vision/LICENSE
36
+ 4683bbfa3df4e757e85ce205c4c078a5c2e7521f8f50fc6d596fd7f65e6fa198 vision/README.md
37
+ 8328816e999c1fec44601f38e9063505b67f12541736f136e53f94f5a10fb783 vision/__init__.py
38
+ 72d2920650ee35d0814aeb3b7450431f56b132a2d75d72ff9a0a9df6ea0203f6 vision/base_vision_provenance.json
39
+ cd7e84e77580254ad9e898a962f1e09c55170714f2a7b81d7025047e60200aa7 vision/example.py
40
+ 371c4962ed7901a4629058eacf3984993b1c871ff10c49e2cc8b0f52485fec6b vision/example_request.json
41
+ f0a8d8182bb654ea829bf37415b1b4ef72272546f69dda2d8f9cfbef921131c4 vision/http_example.py
42
+ b376c208ba3647173ab029f52e34aac5a783460bd06a643db401a1422fd02422 vision/predictor.py
43
+ 60127324a1cab22c6e8f1314efcb7ea736dc038420c63f34e4ee527da1e62363 vision/verification.json
assets/.gitkeep ADDED
File without changes
assets/jev-six-demo-grid.gif ADDED

Git LFS Details

  • SHA256: 9d0b6dfa93abc37c50872b9e568b1c06c73ca867bb4592f92429691cca83bebc
  • Pointer size: 133 Bytes
  • Size of remote file: 23.6 MB
assets/jev-snake-demo.gif ADDED

Git LFS Details

  • SHA256: 3e5547987b4976248251215d90ae234772988a2d3099bdc77c48dc70ba117b0b
  • Pointer size: 133 Bytes
  • Size of remote file: 17.6 MB
backbone/config.json ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5Model"
4
+ ],
5
+ "image_token_id": 248056,
6
+ "model_type": "qwen3_5",
7
+ "text_config": {
8
+ "architectures": [
9
+ "Qwen3_5TextModel"
10
+ ],
11
+ "attention_bias": false,
12
+ "attention_dropout": 0.0,
13
+ "attn_output_gate": true,
14
+ "bos_token_id": null,
15
+ "dtype": "bfloat16",
16
+ "eos_token_id": 248044,
17
+ "full_attention_interval": 4,
18
+ "head_dim": 256,
19
+ "hidden_act": "silu",
20
+ "hidden_size": 2560,
21
+ "initializer_range": 0.02,
22
+ "intermediate_size": 9216,
23
+ "layer_types": [
24
+ "linear_attention",
25
+ "linear_attention",
26
+ "linear_attention",
27
+ "full_attention",
28
+ "linear_attention",
29
+ "linear_attention",
30
+ "linear_attention",
31
+ "full_attention",
32
+ "linear_attention",
33
+ "linear_attention",
34
+ "linear_attention",
35
+ "full_attention",
36
+ "linear_attention",
37
+ "linear_attention",
38
+ "linear_attention",
39
+ "full_attention",
40
+ "linear_attention",
41
+ "linear_attention",
42
+ "linear_attention",
43
+ "full_attention",
44
+ "linear_attention",
45
+ "linear_attention",
46
+ "linear_attention",
47
+ "full_attention",
48
+ "linear_attention",
49
+ "linear_attention",
50
+ "linear_attention",
51
+ "full_attention",
52
+ "linear_attention",
53
+ "linear_attention",
54
+ "linear_attention",
55
+ "full_attention"
56
+ ],
57
+ "linear_conv_kernel_dim": 4,
58
+ "linear_key_head_dim": 128,
59
+ "linear_num_key_heads": 16,
60
+ "linear_num_value_heads": 32,
61
+ "linear_value_head_dim": 128,
62
+ "mamba_ssm_dtype": "float32",
63
+ "max_position_embeddings": 262144,
64
+ "mlp_only_layers": [],
65
+ "model_type": "qwen3_5_text",
66
+ "modification_notice": "Modified by TokenRhythm: language-model weights fine-tuned from Qwen/Qwen3.5-4B; repackaged for text-only inference by changing config and tensor key prefixes. Tensor values are unchanged by repackaging. Original model: Copyright 2026 Alibaba Cloud.",
67
+ "mtp_num_hidden_layers": 1,
68
+ "mtp_use_dedicated_embeddings": false,
69
+ "num_attention_heads": 16,
70
+ "num_hidden_layers": 32,
71
+ "num_key_value_heads": 4,
72
+ "pad_token_id": null,
73
+ "partial_rotary_factor": 0.25,
74
+ "rms_norm_eps": 1e-06,
75
+ "rope_parameters": {
76
+ "mrope_interleaved": true,
77
+ "mrope_section": [
78
+ 11,
79
+ 11,
80
+ 10
81
+ ],
82
+ "partial_rotary_factor": 0.25,
83
+ "rope_theta": 10000000,
84
+ "rope_type": "default"
85
+ },
86
+ "tie_word_embeddings": true,
87
+ "transformers_version": "5.17.0",
88
+ "use_cache": true,
89
+ "vocab_size": 248320
90
+ },
91
+ "tie_word_embeddings": true,
92
+ "transformers_version": "5.17.0",
93
+ "video_token_id": 248057,
94
+ "vision_config": {
95
+ "deepstack_visual_indexes": [],
96
+ "depth": 24,
97
+ "hidden_act": "gelu_pytorch_tanh",
98
+ "hidden_size": 1024,
99
+ "in_channels": 3,
100
+ "initializer_range": 0.02,
101
+ "intermediate_size": 4096,
102
+ "model_type": "qwen3_5",
103
+ "num_heads": 16,
104
+ "num_position_embeddings": 2304,
105
+ "out_hidden_size": 2560,
106
+ "patch_size": 16,
107
+ "spatial_merge_size": 2,
108
+ "temporal_patch_size": 2
109
+ },
110
+ "vision_end_token_id": 248054,
111
+ "vision_start_token_id": 248053
112
+ }
backbone/model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7d1adbb748ff60a91b3b6ffba1ff70bfcab855ff2b8cf5e33f9c6bc1ff13cb7e
3
+ size 3991297968
backbone/model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c37c278c3977b16b7358421a10b5805615a0f393f19f56b540644dded4d0e6c3
3
+ size 3968952928
backbone/model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:475b9a4b012cc888da1d6575746d2b329a944a6107f8e20b1dcb83b116bf1b1d
3
+ size 1118364688
backbone/model.safetensors.index.json ADDED
@@ -0,0 +1,730 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 9078531072
4
+ },
5
+ "weight_map": {
6
+ "language_model.embed_tokens.weight": "model-00001-of-00003.safetensors",
7
+ "language_model.layers.0.input_layernorm.weight": "model-00001-of-00003.safetensors",
8
+ "language_model.layers.0.linear_attn.A_log": "model-00001-of-00003.safetensors",
9
+ "language_model.layers.0.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
10
+ "language_model.layers.0.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
11
+ "language_model.layers.0.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
12
+ "language_model.layers.0.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
13
+ "language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
14
+ "language_model.layers.0.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
15
+ "language_model.layers.0.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
16
+ "language_model.layers.0.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
17
+ "language_model.layers.0.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
18
+ "language_model.layers.0.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
19
+ "language_model.layers.0.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
20
+ "language_model.layers.0.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
21
+ "language_model.layers.1.input_layernorm.weight": "model-00001-of-00003.safetensors",
22
+ "language_model.layers.1.linear_attn.A_log": "model-00001-of-00003.safetensors",
23
+ "language_model.layers.1.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
24
+ "language_model.layers.1.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
25
+ "language_model.layers.1.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
26
+ "language_model.layers.1.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
27
+ "language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
28
+ "language_model.layers.1.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
29
+ "language_model.layers.1.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
30
+ "language_model.layers.1.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
31
+ "language_model.layers.1.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
32
+ "language_model.layers.1.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
33
+ "language_model.layers.1.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
34
+ "language_model.layers.1.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
35
+ "language_model.layers.10.input_layernorm.weight": "model-00001-of-00003.safetensors",
36
+ "language_model.layers.10.linear_attn.A_log": "model-00001-of-00003.safetensors",
37
+ "language_model.layers.10.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
38
+ "language_model.layers.10.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
39
+ "language_model.layers.10.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
40
+ "language_model.layers.10.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
41
+ "language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
42
+ "language_model.layers.10.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
43
+ "language_model.layers.10.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
44
+ "language_model.layers.10.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
45
+ "language_model.layers.10.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
46
+ "language_model.layers.10.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
47
+ "language_model.layers.10.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
48
+ "language_model.layers.10.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
49
+ "language_model.layers.11.input_layernorm.weight": "model-00001-of-00003.safetensors",
50
+ "language_model.layers.11.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
51
+ "language_model.layers.11.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
52
+ "language_model.layers.11.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
53
+ "language_model.layers.11.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
54
+ "language_model.layers.11.self_attn.k_norm.weight": "model-00001-of-00003.safetensors",
55
+ "language_model.layers.11.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
56
+ "language_model.layers.11.self_attn.o_proj.weight": "model-00001-of-00003.safetensors",
57
+ "language_model.layers.11.self_attn.q_norm.weight": "model-00001-of-00003.safetensors",
58
+ "language_model.layers.11.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
59
+ "language_model.layers.11.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
60
+ "language_model.layers.12.input_layernorm.weight": "model-00001-of-00003.safetensors",
61
+ "language_model.layers.12.linear_attn.A_log": "model-00001-of-00003.safetensors",
62
+ "language_model.layers.12.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
63
+ "language_model.layers.12.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
64
+ "language_model.layers.12.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
65
+ "language_model.layers.12.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
66
+ "language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
67
+ "language_model.layers.12.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
68
+ "language_model.layers.12.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
69
+ "language_model.layers.12.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
70
+ "language_model.layers.12.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
71
+ "language_model.layers.12.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
72
+ "language_model.layers.12.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
73
+ "language_model.layers.12.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
74
+ "language_model.layers.13.input_layernorm.weight": "model-00001-of-00003.safetensors",
75
+ "language_model.layers.13.linear_attn.A_log": "model-00001-of-00003.safetensors",
76
+ "language_model.layers.13.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
77
+ "language_model.layers.13.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
78
+ "language_model.layers.13.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
79
+ "language_model.layers.13.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
80
+ "language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
81
+ "language_model.layers.13.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
82
+ "language_model.layers.13.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
83
+ "language_model.layers.13.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
84
+ "language_model.layers.13.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
85
+ "language_model.layers.13.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
86
+ "language_model.layers.13.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
87
+ "language_model.layers.13.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
88
+ "language_model.layers.14.input_layernorm.weight": "model-00001-of-00003.safetensors",
89
+ "language_model.layers.14.linear_attn.A_log": "model-00001-of-00003.safetensors",
90
+ "language_model.layers.14.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
91
+ "language_model.layers.14.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
92
+ "language_model.layers.14.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
93
+ "language_model.layers.14.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
94
+ "language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
95
+ "language_model.layers.14.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
96
+ "language_model.layers.14.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
97
+ "language_model.layers.14.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
98
+ "language_model.layers.14.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
99
+ "language_model.layers.14.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
100
+ "language_model.layers.14.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
101
+ "language_model.layers.14.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
102
+ "language_model.layers.15.input_layernorm.weight": "model-00001-of-00003.safetensors",
103
+ "language_model.layers.15.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
104
+ "language_model.layers.15.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
105
+ "language_model.layers.15.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
106
+ "language_model.layers.15.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
107
+ "language_model.layers.15.self_attn.k_norm.weight": "model-00001-of-00003.safetensors",
108
+ "language_model.layers.15.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
109
+ "language_model.layers.15.self_attn.o_proj.weight": "model-00001-of-00003.safetensors",
110
+ "language_model.layers.15.self_attn.q_norm.weight": "model-00001-of-00003.safetensors",
111
+ "language_model.layers.15.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
112
+ "language_model.layers.15.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
113
+ "language_model.layers.16.input_layernorm.weight": "model-00001-of-00003.safetensors",
114
+ "language_model.layers.16.linear_attn.A_log": "model-00001-of-00003.safetensors",
115
+ "language_model.layers.16.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
116
+ "language_model.layers.16.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
117
+ "language_model.layers.16.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
118
+ "language_model.layers.16.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
119
+ "language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
120
+ "language_model.layers.16.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
121
+ "language_model.layers.16.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
122
+ "language_model.layers.16.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
123
+ "language_model.layers.16.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
124
+ "language_model.layers.16.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
125
+ "language_model.layers.16.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
126
+ "language_model.layers.16.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
127
+ "language_model.layers.17.input_layernorm.weight": "model-00001-of-00003.safetensors",
128
+ "language_model.layers.17.linear_attn.A_log": "model-00001-of-00003.safetensors",
129
+ "language_model.layers.17.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
130
+ "language_model.layers.17.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
131
+ "language_model.layers.17.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
132
+ "language_model.layers.17.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
133
+ "language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
134
+ "language_model.layers.17.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
135
+ "language_model.layers.17.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
136
+ "language_model.layers.17.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
137
+ "language_model.layers.17.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
138
+ "language_model.layers.17.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
139
+ "language_model.layers.17.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
140
+ "language_model.layers.17.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
141
+ "language_model.layers.18.input_layernorm.weight": "model-00001-of-00003.safetensors",
142
+ "language_model.layers.18.linear_attn.A_log": "model-00001-of-00003.safetensors",
143
+ "language_model.layers.18.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
144
+ "language_model.layers.18.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
145
+ "language_model.layers.18.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
146
+ "language_model.layers.18.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
147
+ "language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
148
+ "language_model.layers.18.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors",
149
+ "language_model.layers.18.linear_attn.norm.weight": "model-00001-of-00003.safetensors",
150
+ "language_model.layers.18.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors",
151
+ "language_model.layers.18.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
152
+ "language_model.layers.18.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
153
+ "language_model.layers.18.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
154
+ "language_model.layers.18.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
155
+ "language_model.layers.19.input_layernorm.weight": "model-00001-of-00003.safetensors",
156
+ "language_model.layers.19.mlp.down_proj.weight": "model-00001-of-00003.safetensors",
157
+ "language_model.layers.19.mlp.gate_proj.weight": "model-00001-of-00003.safetensors",
158
+ "language_model.layers.19.mlp.up_proj.weight": "model-00001-of-00003.safetensors",
159
+ "language_model.layers.19.post_attention_layernorm.weight": "model-00001-of-00003.safetensors",
160
+ "language_model.layers.19.self_attn.k_norm.weight": "model-00001-of-00003.safetensors",
161
+ "language_model.layers.19.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
162
+ "language_model.layers.19.self_attn.o_proj.weight": "model-00001-of-00003.safetensors",
163
+ "language_model.layers.19.self_attn.q_norm.weight": "model-00001-of-00003.safetensors",
164
+ "language_model.layers.19.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
165
+ "language_model.layers.19.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
166
+ "language_model.layers.2.input_layernorm.weight": "model-00001-of-00003.safetensors",
167
+ "language_model.layers.2.linear_attn.A_log": "model-00001-of-00003.safetensors",
168
+ "language_model.layers.2.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors",
169
+ "language_model.layers.2.linear_attn.dt_bias": "model-00001-of-00003.safetensors",
170
+ "language_model.layers.2.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors",
171
+ "language_model.layers.2.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors",
172
+ "language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors",
173
+ "language_model.layers.2.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
174
+ "language_model.layers.2.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
175
+ "language_model.layers.2.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
176
+ "language_model.layers.2.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
177
+ "language_model.layers.2.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
178
+ "language_model.layers.2.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
179
+ "language_model.layers.2.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
180
+ "language_model.layers.20.input_layernorm.weight": "model-00002-of-00003.safetensors",
181
+ "language_model.layers.20.linear_attn.A_log": "model-00002-of-00003.safetensors",
182
+ "language_model.layers.20.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
183
+ "language_model.layers.20.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
184
+ "language_model.layers.20.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
185
+ "language_model.layers.20.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
186
+ "language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
187
+ "language_model.layers.20.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
188
+ "language_model.layers.20.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
189
+ "language_model.layers.20.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
190
+ "language_model.layers.20.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
191
+ "language_model.layers.20.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
192
+ "language_model.layers.20.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
193
+ "language_model.layers.20.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
194
+ "language_model.layers.21.input_layernorm.weight": "model-00002-of-00003.safetensors",
195
+ "language_model.layers.21.linear_attn.A_log": "model-00002-of-00003.safetensors",
196
+ "language_model.layers.21.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
197
+ "language_model.layers.21.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
198
+ "language_model.layers.21.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
199
+ "language_model.layers.21.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
200
+ "language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
201
+ "language_model.layers.21.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
202
+ "language_model.layers.21.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
203
+ "language_model.layers.21.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
204
+ "language_model.layers.21.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
205
+ "language_model.layers.21.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
206
+ "language_model.layers.21.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
207
+ "language_model.layers.21.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
208
+ "language_model.layers.22.input_layernorm.weight": "model-00002-of-00003.safetensors",
209
+ "language_model.layers.22.linear_attn.A_log": "model-00002-of-00003.safetensors",
210
+ "language_model.layers.22.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
211
+ "language_model.layers.22.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
212
+ "language_model.layers.22.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
213
+ "language_model.layers.22.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
214
+ "language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
215
+ "language_model.layers.22.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
216
+ "language_model.layers.22.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
217
+ "language_model.layers.22.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
218
+ "language_model.layers.22.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
219
+ "language_model.layers.22.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
220
+ "language_model.layers.22.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
221
+ "language_model.layers.22.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
222
+ "language_model.layers.23.input_layernorm.weight": "model-00002-of-00003.safetensors",
223
+ "language_model.layers.23.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
224
+ "language_model.layers.23.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
225
+ "language_model.layers.23.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
226
+ "language_model.layers.23.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
227
+ "language_model.layers.23.self_attn.k_norm.weight": "model-00002-of-00003.safetensors",
228
+ "language_model.layers.23.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
229
+ "language_model.layers.23.self_attn.o_proj.weight": "model-00002-of-00003.safetensors",
230
+ "language_model.layers.23.self_attn.q_norm.weight": "model-00002-of-00003.safetensors",
231
+ "language_model.layers.23.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
232
+ "language_model.layers.23.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
233
+ "language_model.layers.24.input_layernorm.weight": "model-00002-of-00003.safetensors",
234
+ "language_model.layers.24.linear_attn.A_log": "model-00002-of-00003.safetensors",
235
+ "language_model.layers.24.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
236
+ "language_model.layers.24.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
237
+ "language_model.layers.24.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
238
+ "language_model.layers.24.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
239
+ "language_model.layers.24.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
240
+ "language_model.layers.24.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
241
+ "language_model.layers.24.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
242
+ "language_model.layers.24.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
243
+ "language_model.layers.24.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
244
+ "language_model.layers.24.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
245
+ "language_model.layers.24.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
246
+ "language_model.layers.24.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
247
+ "language_model.layers.25.input_layernorm.weight": "model-00002-of-00003.safetensors",
248
+ "language_model.layers.25.linear_attn.A_log": "model-00002-of-00003.safetensors",
249
+ "language_model.layers.25.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
250
+ "language_model.layers.25.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
251
+ "language_model.layers.25.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
252
+ "language_model.layers.25.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
253
+ "language_model.layers.25.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
254
+ "language_model.layers.25.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
255
+ "language_model.layers.25.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
256
+ "language_model.layers.25.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
257
+ "language_model.layers.25.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
258
+ "language_model.layers.25.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
259
+ "language_model.layers.25.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
260
+ "language_model.layers.25.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
261
+ "language_model.layers.26.input_layernorm.weight": "model-00002-of-00003.safetensors",
262
+ "language_model.layers.26.linear_attn.A_log": "model-00002-of-00003.safetensors",
263
+ "language_model.layers.26.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
264
+ "language_model.layers.26.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
265
+ "language_model.layers.26.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
266
+ "language_model.layers.26.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
267
+ "language_model.layers.26.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
268
+ "language_model.layers.26.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
269
+ "language_model.layers.26.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
270
+ "language_model.layers.26.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
271
+ "language_model.layers.26.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
272
+ "language_model.layers.26.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
273
+ "language_model.layers.26.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
274
+ "language_model.layers.26.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
275
+ "language_model.layers.27.input_layernorm.weight": "model-00002-of-00003.safetensors",
276
+ "language_model.layers.27.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
277
+ "language_model.layers.27.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
278
+ "language_model.layers.27.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
279
+ "language_model.layers.27.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
280
+ "language_model.layers.27.self_attn.k_norm.weight": "model-00002-of-00003.safetensors",
281
+ "language_model.layers.27.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
282
+ "language_model.layers.27.self_attn.o_proj.weight": "model-00002-of-00003.safetensors",
283
+ "language_model.layers.27.self_attn.q_norm.weight": "model-00002-of-00003.safetensors",
284
+ "language_model.layers.27.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
285
+ "language_model.layers.27.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
286
+ "language_model.layers.28.input_layernorm.weight": "model-00002-of-00003.safetensors",
287
+ "language_model.layers.28.linear_attn.A_log": "model-00002-of-00003.safetensors",
288
+ "language_model.layers.28.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
289
+ "language_model.layers.28.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
290
+ "language_model.layers.28.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
291
+ "language_model.layers.28.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
292
+ "language_model.layers.28.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
293
+ "language_model.layers.28.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
294
+ "language_model.layers.28.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
295
+ "language_model.layers.28.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
296
+ "language_model.layers.28.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
297
+ "language_model.layers.28.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
298
+ "language_model.layers.28.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
299
+ "language_model.layers.28.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
300
+ "language_model.layers.29.input_layernorm.weight": "model-00002-of-00003.safetensors",
301
+ "language_model.layers.29.linear_attn.A_log": "model-00002-of-00003.safetensors",
302
+ "language_model.layers.29.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
303
+ "language_model.layers.29.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
304
+ "language_model.layers.29.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
305
+ "language_model.layers.29.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
306
+ "language_model.layers.29.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
307
+ "language_model.layers.29.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
308
+ "language_model.layers.29.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
309
+ "language_model.layers.29.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
310
+ "language_model.layers.29.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
311
+ "language_model.layers.29.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
312
+ "language_model.layers.29.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
313
+ "language_model.layers.29.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
314
+ "language_model.layers.3.input_layernorm.weight": "model-00002-of-00003.safetensors",
315
+ "language_model.layers.3.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
316
+ "language_model.layers.3.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
317
+ "language_model.layers.3.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
318
+ "language_model.layers.3.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
319
+ "language_model.layers.3.self_attn.k_norm.weight": "model-00002-of-00003.safetensors",
320
+ "language_model.layers.3.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
321
+ "language_model.layers.3.self_attn.o_proj.weight": "model-00002-of-00003.safetensors",
322
+ "language_model.layers.3.self_attn.q_norm.weight": "model-00002-of-00003.safetensors",
323
+ "language_model.layers.3.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
324
+ "language_model.layers.3.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
325
+ "language_model.layers.30.input_layernorm.weight": "model-00002-of-00003.safetensors",
326
+ "language_model.layers.30.linear_attn.A_log": "model-00002-of-00003.safetensors",
327
+ "language_model.layers.30.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
328
+ "language_model.layers.30.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
329
+ "language_model.layers.30.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
330
+ "language_model.layers.30.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
331
+ "language_model.layers.30.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
332
+ "language_model.layers.30.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
333
+ "language_model.layers.30.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
334
+ "language_model.layers.30.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
335
+ "language_model.layers.30.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
336
+ "language_model.layers.30.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
337
+ "language_model.layers.30.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
338
+ "language_model.layers.30.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
339
+ "language_model.layers.31.input_layernorm.weight": "model-00002-of-00003.safetensors",
340
+ "language_model.layers.31.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
341
+ "language_model.layers.31.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
342
+ "language_model.layers.31.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
343
+ "language_model.layers.31.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
344
+ "language_model.layers.31.self_attn.k_norm.weight": "model-00002-of-00003.safetensors",
345
+ "language_model.layers.31.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
346
+ "language_model.layers.31.self_attn.o_proj.weight": "model-00002-of-00003.safetensors",
347
+ "language_model.layers.31.self_attn.q_norm.weight": "model-00002-of-00003.safetensors",
348
+ "language_model.layers.31.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
349
+ "language_model.layers.31.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
350
+ "language_model.layers.4.input_layernorm.weight": "model-00002-of-00003.safetensors",
351
+ "language_model.layers.4.linear_attn.A_log": "model-00002-of-00003.safetensors",
352
+ "language_model.layers.4.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
353
+ "language_model.layers.4.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
354
+ "language_model.layers.4.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
355
+ "language_model.layers.4.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
356
+ "language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
357
+ "language_model.layers.4.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
358
+ "language_model.layers.4.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
359
+ "language_model.layers.4.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
360
+ "language_model.layers.4.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
361
+ "language_model.layers.4.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
362
+ "language_model.layers.4.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
363
+ "language_model.layers.4.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
364
+ "language_model.layers.5.input_layernorm.weight": "model-00002-of-00003.safetensors",
365
+ "language_model.layers.5.linear_attn.A_log": "model-00002-of-00003.safetensors",
366
+ "language_model.layers.5.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
367
+ "language_model.layers.5.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
368
+ "language_model.layers.5.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
369
+ "language_model.layers.5.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
370
+ "language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
371
+ "language_model.layers.5.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
372
+ "language_model.layers.5.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
373
+ "language_model.layers.5.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
374
+ "language_model.layers.5.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
375
+ "language_model.layers.5.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
376
+ "language_model.layers.5.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
377
+ "language_model.layers.5.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
378
+ "language_model.layers.6.input_layernorm.weight": "model-00002-of-00003.safetensors",
379
+ "language_model.layers.6.linear_attn.A_log": "model-00002-of-00003.safetensors",
380
+ "language_model.layers.6.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
381
+ "language_model.layers.6.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
382
+ "language_model.layers.6.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
383
+ "language_model.layers.6.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
384
+ "language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors",
385
+ "language_model.layers.6.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors",
386
+ "language_model.layers.6.linear_attn.norm.weight": "model-00002-of-00003.safetensors",
387
+ "language_model.layers.6.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors",
388
+ "language_model.layers.6.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
389
+ "language_model.layers.6.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
390
+ "language_model.layers.6.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
391
+ "language_model.layers.6.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
392
+ "language_model.layers.7.input_layernorm.weight": "model-00002-of-00003.safetensors",
393
+ "language_model.layers.7.mlp.down_proj.weight": "model-00002-of-00003.safetensors",
394
+ "language_model.layers.7.mlp.gate_proj.weight": "model-00002-of-00003.safetensors",
395
+ "language_model.layers.7.mlp.up_proj.weight": "model-00002-of-00003.safetensors",
396
+ "language_model.layers.7.post_attention_layernorm.weight": "model-00002-of-00003.safetensors",
397
+ "language_model.layers.7.self_attn.k_norm.weight": "model-00002-of-00003.safetensors",
398
+ "language_model.layers.7.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
399
+ "language_model.layers.7.self_attn.o_proj.weight": "model-00002-of-00003.safetensors",
400
+ "language_model.layers.7.self_attn.q_norm.weight": "model-00002-of-00003.safetensors",
401
+ "language_model.layers.7.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
402
+ "language_model.layers.7.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
403
+ "language_model.layers.8.input_layernorm.weight": "model-00002-of-00003.safetensors",
404
+ "language_model.layers.8.linear_attn.A_log": "model-00002-of-00003.safetensors",
405
+ "language_model.layers.8.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors",
406
+ "language_model.layers.8.linear_attn.dt_bias": "model-00002-of-00003.safetensors",
407
+ "language_model.layers.8.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors",
408
+ "language_model.layers.8.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors",
409
+ "language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00003-of-00003.safetensors",
410
+ "language_model.layers.8.linear_attn.in_proj_z.weight": "model-00003-of-00003.safetensors",
411
+ "language_model.layers.8.linear_attn.norm.weight": "model-00003-of-00003.safetensors",
412
+ "language_model.layers.8.linear_attn.out_proj.weight": "model-00003-of-00003.safetensors",
413
+ "language_model.layers.8.mlp.down_proj.weight": "model-00003-of-00003.safetensors",
414
+ "language_model.layers.8.mlp.gate_proj.weight": "model-00003-of-00003.safetensors",
415
+ "language_model.layers.8.mlp.up_proj.weight": "model-00003-of-00003.safetensors",
416
+ "language_model.layers.8.post_attention_layernorm.weight": "model-00003-of-00003.safetensors",
417
+ "language_model.layers.9.input_layernorm.weight": "model-00003-of-00003.safetensors",
418
+ "language_model.layers.9.linear_attn.A_log": "model-00003-of-00003.safetensors",
419
+ "language_model.layers.9.linear_attn.conv1d.weight": "model-00003-of-00003.safetensors",
420
+ "language_model.layers.9.linear_attn.dt_bias": "model-00003-of-00003.safetensors",
421
+ "language_model.layers.9.linear_attn.in_proj_a.weight": "model-00003-of-00003.safetensors",
422
+ "language_model.layers.9.linear_attn.in_proj_b.weight": "model-00003-of-00003.safetensors",
423
+ "language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00003-of-00003.safetensors",
424
+ "language_model.layers.9.linear_attn.in_proj_z.weight": "model-00003-of-00003.safetensors",
425
+ "language_model.layers.9.linear_attn.norm.weight": "model-00003-of-00003.safetensors",
426
+ "language_model.layers.9.linear_attn.out_proj.weight": "model-00003-of-00003.safetensors",
427
+ "language_model.layers.9.mlp.down_proj.weight": "model-00003-of-00003.safetensors",
428
+ "language_model.layers.9.mlp.gate_proj.weight": "model-00003-of-00003.safetensors",
429
+ "language_model.layers.9.mlp.up_proj.weight": "model-00003-of-00003.safetensors",
430
+ "language_model.layers.9.post_attention_layernorm.weight": "model-00003-of-00003.safetensors",
431
+ "language_model.norm.weight": "model-00003-of-00003.safetensors",
432
+ "visual.blocks.0.attn.proj.bias": "model-00003-of-00003.safetensors",
433
+ "visual.blocks.0.attn.proj.weight": "model-00003-of-00003.safetensors",
434
+ "visual.blocks.0.attn.qkv.bias": "model-00003-of-00003.safetensors",
435
+ "visual.blocks.0.attn.qkv.weight": "model-00003-of-00003.safetensors",
436
+ "visual.blocks.0.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
437
+ "visual.blocks.0.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
438
+ "visual.blocks.0.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
439
+ "visual.blocks.0.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
440
+ "visual.blocks.0.norm1.bias": "model-00003-of-00003.safetensors",
441
+ "visual.blocks.0.norm1.weight": "model-00003-of-00003.safetensors",
442
+ "visual.blocks.0.norm2.bias": "model-00003-of-00003.safetensors",
443
+ "visual.blocks.0.norm2.weight": "model-00003-of-00003.safetensors",
444
+ "visual.blocks.1.attn.proj.bias": "model-00003-of-00003.safetensors",
445
+ "visual.blocks.1.attn.proj.weight": "model-00003-of-00003.safetensors",
446
+ "visual.blocks.1.attn.qkv.bias": "model-00003-of-00003.safetensors",
447
+ "visual.blocks.1.attn.qkv.weight": "model-00003-of-00003.safetensors",
448
+ "visual.blocks.1.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
449
+ "visual.blocks.1.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
450
+ "visual.blocks.1.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
451
+ "visual.blocks.1.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
452
+ "visual.blocks.1.norm1.bias": "model-00003-of-00003.safetensors",
453
+ "visual.blocks.1.norm1.weight": "model-00003-of-00003.safetensors",
454
+ "visual.blocks.1.norm2.bias": "model-00003-of-00003.safetensors",
455
+ "visual.blocks.1.norm2.weight": "model-00003-of-00003.safetensors",
456
+ "visual.blocks.10.attn.proj.bias": "model-00003-of-00003.safetensors",
457
+ "visual.blocks.10.attn.proj.weight": "model-00003-of-00003.safetensors",
458
+ "visual.blocks.10.attn.qkv.bias": "model-00003-of-00003.safetensors",
459
+ "visual.blocks.10.attn.qkv.weight": "model-00003-of-00003.safetensors",
460
+ "visual.blocks.10.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
461
+ "visual.blocks.10.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
462
+ "visual.blocks.10.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
463
+ "visual.blocks.10.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
464
+ "visual.blocks.10.norm1.bias": "model-00003-of-00003.safetensors",
465
+ "visual.blocks.10.norm1.weight": "model-00003-of-00003.safetensors",
466
+ "visual.blocks.10.norm2.bias": "model-00003-of-00003.safetensors",
467
+ "visual.blocks.10.norm2.weight": "model-00003-of-00003.safetensors",
468
+ "visual.blocks.11.attn.proj.bias": "model-00003-of-00003.safetensors",
469
+ "visual.blocks.11.attn.proj.weight": "model-00003-of-00003.safetensors",
470
+ "visual.blocks.11.attn.qkv.bias": "model-00003-of-00003.safetensors",
471
+ "visual.blocks.11.attn.qkv.weight": "model-00003-of-00003.safetensors",
472
+ "visual.blocks.11.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
473
+ "visual.blocks.11.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
474
+ "visual.blocks.11.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
475
+ "visual.blocks.11.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
476
+ "visual.blocks.11.norm1.bias": "model-00003-of-00003.safetensors",
477
+ "visual.blocks.11.norm1.weight": "model-00003-of-00003.safetensors",
478
+ "visual.blocks.11.norm2.bias": "model-00003-of-00003.safetensors",
479
+ "visual.blocks.11.norm2.weight": "model-00003-of-00003.safetensors",
480
+ "visual.blocks.12.attn.proj.bias": "model-00003-of-00003.safetensors",
481
+ "visual.blocks.12.attn.proj.weight": "model-00003-of-00003.safetensors",
482
+ "visual.blocks.12.attn.qkv.bias": "model-00003-of-00003.safetensors",
483
+ "visual.blocks.12.attn.qkv.weight": "model-00003-of-00003.safetensors",
484
+ "visual.blocks.12.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
485
+ "visual.blocks.12.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
486
+ "visual.blocks.12.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
487
+ "visual.blocks.12.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
488
+ "visual.blocks.12.norm1.bias": "model-00003-of-00003.safetensors",
489
+ "visual.blocks.12.norm1.weight": "model-00003-of-00003.safetensors",
490
+ "visual.blocks.12.norm2.bias": "model-00003-of-00003.safetensors",
491
+ "visual.blocks.12.norm2.weight": "model-00003-of-00003.safetensors",
492
+ "visual.blocks.13.attn.proj.bias": "model-00003-of-00003.safetensors",
493
+ "visual.blocks.13.attn.proj.weight": "model-00003-of-00003.safetensors",
494
+ "visual.blocks.13.attn.qkv.bias": "model-00003-of-00003.safetensors",
495
+ "visual.blocks.13.attn.qkv.weight": "model-00003-of-00003.safetensors",
496
+ "visual.blocks.13.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
497
+ "visual.blocks.13.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
498
+ "visual.blocks.13.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
499
+ "visual.blocks.13.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
500
+ "visual.blocks.13.norm1.bias": "model-00003-of-00003.safetensors",
501
+ "visual.blocks.13.norm1.weight": "model-00003-of-00003.safetensors",
502
+ "visual.blocks.13.norm2.bias": "model-00003-of-00003.safetensors",
503
+ "visual.blocks.13.norm2.weight": "model-00003-of-00003.safetensors",
504
+ "visual.blocks.14.attn.proj.bias": "model-00003-of-00003.safetensors",
505
+ "visual.blocks.14.attn.proj.weight": "model-00003-of-00003.safetensors",
506
+ "visual.blocks.14.attn.qkv.bias": "model-00003-of-00003.safetensors",
507
+ "visual.blocks.14.attn.qkv.weight": "model-00003-of-00003.safetensors",
508
+ "visual.blocks.14.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
509
+ "visual.blocks.14.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
510
+ "visual.blocks.14.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
511
+ "visual.blocks.14.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
512
+ "visual.blocks.14.norm1.bias": "model-00003-of-00003.safetensors",
513
+ "visual.blocks.14.norm1.weight": "model-00003-of-00003.safetensors",
514
+ "visual.blocks.14.norm2.bias": "model-00003-of-00003.safetensors",
515
+ "visual.blocks.14.norm2.weight": "model-00003-of-00003.safetensors",
516
+ "visual.blocks.15.attn.proj.bias": "model-00003-of-00003.safetensors",
517
+ "visual.blocks.15.attn.proj.weight": "model-00003-of-00003.safetensors",
518
+ "visual.blocks.15.attn.qkv.bias": "model-00003-of-00003.safetensors",
519
+ "visual.blocks.15.attn.qkv.weight": "model-00003-of-00003.safetensors",
520
+ "visual.blocks.15.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
521
+ "visual.blocks.15.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
522
+ "visual.blocks.15.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
523
+ "visual.blocks.15.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
524
+ "visual.blocks.15.norm1.bias": "model-00003-of-00003.safetensors",
525
+ "visual.blocks.15.norm1.weight": "model-00003-of-00003.safetensors",
526
+ "visual.blocks.15.norm2.bias": "model-00003-of-00003.safetensors",
527
+ "visual.blocks.15.norm2.weight": "model-00003-of-00003.safetensors",
528
+ "visual.blocks.16.attn.proj.bias": "model-00003-of-00003.safetensors",
529
+ "visual.blocks.16.attn.proj.weight": "model-00003-of-00003.safetensors",
530
+ "visual.blocks.16.attn.qkv.bias": "model-00003-of-00003.safetensors",
531
+ "visual.blocks.16.attn.qkv.weight": "model-00003-of-00003.safetensors",
532
+ "visual.blocks.16.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
533
+ "visual.blocks.16.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
534
+ "visual.blocks.16.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
535
+ "visual.blocks.16.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
536
+ "visual.blocks.16.norm1.bias": "model-00003-of-00003.safetensors",
537
+ "visual.blocks.16.norm1.weight": "model-00003-of-00003.safetensors",
538
+ "visual.blocks.16.norm2.bias": "model-00003-of-00003.safetensors",
539
+ "visual.blocks.16.norm2.weight": "model-00003-of-00003.safetensors",
540
+ "visual.blocks.17.attn.proj.bias": "model-00003-of-00003.safetensors",
541
+ "visual.blocks.17.attn.proj.weight": "model-00003-of-00003.safetensors",
542
+ "visual.blocks.17.attn.qkv.bias": "model-00003-of-00003.safetensors",
543
+ "visual.blocks.17.attn.qkv.weight": "model-00003-of-00003.safetensors",
544
+ "visual.blocks.17.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
545
+ "visual.blocks.17.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
546
+ "visual.blocks.17.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
547
+ "visual.blocks.17.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
548
+ "visual.blocks.17.norm1.bias": "model-00003-of-00003.safetensors",
549
+ "visual.blocks.17.norm1.weight": "model-00003-of-00003.safetensors",
550
+ "visual.blocks.17.norm2.bias": "model-00003-of-00003.safetensors",
551
+ "visual.blocks.17.norm2.weight": "model-00003-of-00003.safetensors",
552
+ "visual.blocks.18.attn.proj.bias": "model-00003-of-00003.safetensors",
553
+ "visual.blocks.18.attn.proj.weight": "model-00003-of-00003.safetensors",
554
+ "visual.blocks.18.attn.qkv.bias": "model-00003-of-00003.safetensors",
555
+ "visual.blocks.18.attn.qkv.weight": "model-00003-of-00003.safetensors",
556
+ "visual.blocks.18.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
557
+ "visual.blocks.18.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
558
+ "visual.blocks.18.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
559
+ "visual.blocks.18.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
560
+ "visual.blocks.18.norm1.bias": "model-00003-of-00003.safetensors",
561
+ "visual.blocks.18.norm1.weight": "model-00003-of-00003.safetensors",
562
+ "visual.blocks.18.norm2.bias": "model-00003-of-00003.safetensors",
563
+ "visual.blocks.18.norm2.weight": "model-00003-of-00003.safetensors",
564
+ "visual.blocks.19.attn.proj.bias": "model-00003-of-00003.safetensors",
565
+ "visual.blocks.19.attn.proj.weight": "model-00003-of-00003.safetensors",
566
+ "visual.blocks.19.attn.qkv.bias": "model-00003-of-00003.safetensors",
567
+ "visual.blocks.19.attn.qkv.weight": "model-00003-of-00003.safetensors",
568
+ "visual.blocks.19.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
569
+ "visual.blocks.19.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
570
+ "visual.blocks.19.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
571
+ "visual.blocks.19.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
572
+ "visual.blocks.19.norm1.bias": "model-00003-of-00003.safetensors",
573
+ "visual.blocks.19.norm1.weight": "model-00003-of-00003.safetensors",
574
+ "visual.blocks.19.norm2.bias": "model-00003-of-00003.safetensors",
575
+ "visual.blocks.19.norm2.weight": "model-00003-of-00003.safetensors",
576
+ "visual.blocks.2.attn.proj.bias": "model-00003-of-00003.safetensors",
577
+ "visual.blocks.2.attn.proj.weight": "model-00003-of-00003.safetensors",
578
+ "visual.blocks.2.attn.qkv.bias": "model-00003-of-00003.safetensors",
579
+ "visual.blocks.2.attn.qkv.weight": "model-00003-of-00003.safetensors",
580
+ "visual.blocks.2.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
581
+ "visual.blocks.2.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
582
+ "visual.blocks.2.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
583
+ "visual.blocks.2.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
584
+ "visual.blocks.2.norm1.bias": "model-00003-of-00003.safetensors",
585
+ "visual.blocks.2.norm1.weight": "model-00003-of-00003.safetensors",
586
+ "visual.blocks.2.norm2.bias": "model-00003-of-00003.safetensors",
587
+ "visual.blocks.2.norm2.weight": "model-00003-of-00003.safetensors",
588
+ "visual.blocks.20.attn.proj.bias": "model-00003-of-00003.safetensors",
589
+ "visual.blocks.20.attn.proj.weight": "model-00003-of-00003.safetensors",
590
+ "visual.blocks.20.attn.qkv.bias": "model-00003-of-00003.safetensors",
591
+ "visual.blocks.20.attn.qkv.weight": "model-00003-of-00003.safetensors",
592
+ "visual.blocks.20.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
593
+ "visual.blocks.20.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
594
+ "visual.blocks.20.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
595
+ "visual.blocks.20.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
596
+ "visual.blocks.20.norm1.bias": "model-00003-of-00003.safetensors",
597
+ "visual.blocks.20.norm1.weight": "model-00003-of-00003.safetensors",
598
+ "visual.blocks.20.norm2.bias": "model-00003-of-00003.safetensors",
599
+ "visual.blocks.20.norm2.weight": "model-00003-of-00003.safetensors",
600
+ "visual.blocks.21.attn.proj.bias": "model-00003-of-00003.safetensors",
601
+ "visual.blocks.21.attn.proj.weight": "model-00003-of-00003.safetensors",
602
+ "visual.blocks.21.attn.qkv.bias": "model-00003-of-00003.safetensors",
603
+ "visual.blocks.21.attn.qkv.weight": "model-00003-of-00003.safetensors",
604
+ "visual.blocks.21.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
605
+ "visual.blocks.21.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
606
+ "visual.blocks.21.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
607
+ "visual.blocks.21.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
608
+ "visual.blocks.21.norm1.bias": "model-00003-of-00003.safetensors",
609
+ "visual.blocks.21.norm1.weight": "model-00003-of-00003.safetensors",
610
+ "visual.blocks.21.norm2.bias": "model-00003-of-00003.safetensors",
611
+ "visual.blocks.21.norm2.weight": "model-00003-of-00003.safetensors",
612
+ "visual.blocks.22.attn.proj.bias": "model-00003-of-00003.safetensors",
613
+ "visual.blocks.22.attn.proj.weight": "model-00003-of-00003.safetensors",
614
+ "visual.blocks.22.attn.qkv.bias": "model-00003-of-00003.safetensors",
615
+ "visual.blocks.22.attn.qkv.weight": "model-00003-of-00003.safetensors",
616
+ "visual.blocks.22.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
617
+ "visual.blocks.22.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
618
+ "visual.blocks.22.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
619
+ "visual.blocks.22.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
620
+ "visual.blocks.22.norm1.bias": "model-00003-of-00003.safetensors",
621
+ "visual.blocks.22.norm1.weight": "model-00003-of-00003.safetensors",
622
+ "visual.blocks.22.norm2.bias": "model-00003-of-00003.safetensors",
623
+ "visual.blocks.22.norm2.weight": "model-00003-of-00003.safetensors",
624
+ "visual.blocks.23.attn.proj.bias": "model-00003-of-00003.safetensors",
625
+ "visual.blocks.23.attn.proj.weight": "model-00003-of-00003.safetensors",
626
+ "visual.blocks.23.attn.qkv.bias": "model-00003-of-00003.safetensors",
627
+ "visual.blocks.23.attn.qkv.weight": "model-00003-of-00003.safetensors",
628
+ "visual.blocks.23.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
629
+ "visual.blocks.23.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
630
+ "visual.blocks.23.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
631
+ "visual.blocks.23.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
632
+ "visual.blocks.23.norm1.bias": "model-00003-of-00003.safetensors",
633
+ "visual.blocks.23.norm1.weight": "model-00003-of-00003.safetensors",
634
+ "visual.blocks.23.norm2.bias": "model-00003-of-00003.safetensors",
635
+ "visual.blocks.23.norm2.weight": "model-00003-of-00003.safetensors",
636
+ "visual.blocks.3.attn.proj.bias": "model-00003-of-00003.safetensors",
637
+ "visual.blocks.3.attn.proj.weight": "model-00003-of-00003.safetensors",
638
+ "visual.blocks.3.attn.qkv.bias": "model-00003-of-00003.safetensors",
639
+ "visual.blocks.3.attn.qkv.weight": "model-00003-of-00003.safetensors",
640
+ "visual.blocks.3.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
641
+ "visual.blocks.3.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
642
+ "visual.blocks.3.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
643
+ "visual.blocks.3.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
644
+ "visual.blocks.3.norm1.bias": "model-00003-of-00003.safetensors",
645
+ "visual.blocks.3.norm1.weight": "model-00003-of-00003.safetensors",
646
+ "visual.blocks.3.norm2.bias": "model-00003-of-00003.safetensors",
647
+ "visual.blocks.3.norm2.weight": "model-00003-of-00003.safetensors",
648
+ "visual.blocks.4.attn.proj.bias": "model-00003-of-00003.safetensors",
649
+ "visual.blocks.4.attn.proj.weight": "model-00003-of-00003.safetensors",
650
+ "visual.blocks.4.attn.qkv.bias": "model-00003-of-00003.safetensors",
651
+ "visual.blocks.4.attn.qkv.weight": "model-00003-of-00003.safetensors",
652
+ "visual.blocks.4.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
653
+ "visual.blocks.4.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
654
+ "visual.blocks.4.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
655
+ "visual.blocks.4.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
656
+ "visual.blocks.4.norm1.bias": "model-00003-of-00003.safetensors",
657
+ "visual.blocks.4.norm1.weight": "model-00003-of-00003.safetensors",
658
+ "visual.blocks.4.norm2.bias": "model-00003-of-00003.safetensors",
659
+ "visual.blocks.4.norm2.weight": "model-00003-of-00003.safetensors",
660
+ "visual.blocks.5.attn.proj.bias": "model-00003-of-00003.safetensors",
661
+ "visual.blocks.5.attn.proj.weight": "model-00003-of-00003.safetensors",
662
+ "visual.blocks.5.attn.qkv.bias": "model-00003-of-00003.safetensors",
663
+ "visual.blocks.5.attn.qkv.weight": "model-00003-of-00003.safetensors",
664
+ "visual.blocks.5.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
665
+ "visual.blocks.5.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
666
+ "visual.blocks.5.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
667
+ "visual.blocks.5.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
668
+ "visual.blocks.5.norm1.bias": "model-00003-of-00003.safetensors",
669
+ "visual.blocks.5.norm1.weight": "model-00003-of-00003.safetensors",
670
+ "visual.blocks.5.norm2.bias": "model-00003-of-00003.safetensors",
671
+ "visual.blocks.5.norm2.weight": "model-00003-of-00003.safetensors",
672
+ "visual.blocks.6.attn.proj.bias": "model-00003-of-00003.safetensors",
673
+ "visual.blocks.6.attn.proj.weight": "model-00003-of-00003.safetensors",
674
+ "visual.blocks.6.attn.qkv.bias": "model-00003-of-00003.safetensors",
675
+ "visual.blocks.6.attn.qkv.weight": "model-00003-of-00003.safetensors",
676
+ "visual.blocks.6.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
677
+ "visual.blocks.6.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
678
+ "visual.blocks.6.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
679
+ "visual.blocks.6.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
680
+ "visual.blocks.6.norm1.bias": "model-00003-of-00003.safetensors",
681
+ "visual.blocks.6.norm1.weight": "model-00003-of-00003.safetensors",
682
+ "visual.blocks.6.norm2.bias": "model-00003-of-00003.safetensors",
683
+ "visual.blocks.6.norm2.weight": "model-00003-of-00003.safetensors",
684
+ "visual.blocks.7.attn.proj.bias": "model-00003-of-00003.safetensors",
685
+ "visual.blocks.7.attn.proj.weight": "model-00003-of-00003.safetensors",
686
+ "visual.blocks.7.attn.qkv.bias": "model-00003-of-00003.safetensors",
687
+ "visual.blocks.7.attn.qkv.weight": "model-00003-of-00003.safetensors",
688
+ "visual.blocks.7.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
689
+ "visual.blocks.7.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
690
+ "visual.blocks.7.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
691
+ "visual.blocks.7.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
692
+ "visual.blocks.7.norm1.bias": "model-00003-of-00003.safetensors",
693
+ "visual.blocks.7.norm1.weight": "model-00003-of-00003.safetensors",
694
+ "visual.blocks.7.norm2.bias": "model-00003-of-00003.safetensors",
695
+ "visual.blocks.7.norm2.weight": "model-00003-of-00003.safetensors",
696
+ "visual.blocks.8.attn.proj.bias": "model-00003-of-00003.safetensors",
697
+ "visual.blocks.8.attn.proj.weight": "model-00003-of-00003.safetensors",
698
+ "visual.blocks.8.attn.qkv.bias": "model-00003-of-00003.safetensors",
699
+ "visual.blocks.8.attn.qkv.weight": "model-00003-of-00003.safetensors",
700
+ "visual.blocks.8.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
701
+ "visual.blocks.8.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
702
+ "visual.blocks.8.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
703
+ "visual.blocks.8.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
704
+ "visual.blocks.8.norm1.bias": "model-00003-of-00003.safetensors",
705
+ "visual.blocks.8.norm1.weight": "model-00003-of-00003.safetensors",
706
+ "visual.blocks.8.norm2.bias": "model-00003-of-00003.safetensors",
707
+ "visual.blocks.8.norm2.weight": "model-00003-of-00003.safetensors",
708
+ "visual.blocks.9.attn.proj.bias": "model-00003-of-00003.safetensors",
709
+ "visual.blocks.9.attn.proj.weight": "model-00003-of-00003.safetensors",
710
+ "visual.blocks.9.attn.qkv.bias": "model-00003-of-00003.safetensors",
711
+ "visual.blocks.9.attn.qkv.weight": "model-00003-of-00003.safetensors",
712
+ "visual.blocks.9.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors",
713
+ "visual.blocks.9.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors",
714
+ "visual.blocks.9.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors",
715
+ "visual.blocks.9.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors",
716
+ "visual.blocks.9.norm1.bias": "model-00003-of-00003.safetensors",
717
+ "visual.blocks.9.norm1.weight": "model-00003-of-00003.safetensors",
718
+ "visual.blocks.9.norm2.bias": "model-00003-of-00003.safetensors",
719
+ "visual.blocks.9.norm2.weight": "model-00003-of-00003.safetensors",
720
+ "visual.merger.linear_fc1.bias": "model-00003-of-00003.safetensors",
721
+ "visual.merger.linear_fc1.weight": "model-00003-of-00003.safetensors",
722
+ "visual.merger.linear_fc2.bias": "model-00003-of-00003.safetensors",
723
+ "visual.merger.linear_fc2.weight": "model-00003-of-00003.safetensors",
724
+ "visual.merger.norm.bias": "model-00003-of-00003.safetensors",
725
+ "visual.merger.norm.weight": "model-00003-of-00003.safetensors",
726
+ "visual.patch_embed.proj.bias": "model-00003-of-00003.safetensors",
727
+ "visual.patch_embed.proj.weight": "model-00003-of-00003.safetensors",
728
+ "visual.pos_embed.weight": "model-00003-of-00003.safetensors"
729
+ }
730
+ }
backbone/preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 16777216,
4
+ "shortest_edge": 65536
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "image_processor_type": "Qwen2VLImageProcessorFast"
21
+ }
config.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "NeoHorse-Jev-4B",
3
+ "format": "unified multimodal backbone + independent decision pointer head; NOT chat causal LM",
4
+ "backbone_class": "Qwen3_5Model",
5
+ "backbone_dtype": "bfloat16",
6
+ "head_dtype": "float32",
7
+ "merge_dtype": "float32",
8
+ "lora_scale": 1.0,
9
+ "head_dim": 256,
10
+ "option_isolation": false,
11
+ "hybrid": true,
12
+ "temperature": 1.0,
13
+ "runtime_origin": "https://github.com/jaredpalmer/kev",
14
+ "created_unix": 1790132795.85603,
15
+ "bundle_version": "1.0.0",
16
+ "vision_source": "Qwen/Qwen3.5-4B",
17
+ "vision_tensors": 297,
18
+ "vision_finetuning": false,
19
+ "text_tensors_unchanged": true,
20
+ "pointer_head_unchanged": true,
21
+ "http_image_field": "image: inline PNG/JPEG/WebP base64 data URL"
22
+ }
dist/neohorse_decision-1.0.0-py3-none-any.whl ADDED
Binary file (24.8 kB). View file
 
environment.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "packages": {
3
+ "torch": "2.8.0",
4
+ "transformers": "5.17.0",
5
+ "safetensors": "0.8.0",
6
+ "pydantic": "2.13.5",
7
+ "peft": "0.21.0",
8
+ "triton": "3.7.1",
9
+ "flash-linear-attention": "0.5.2",
10
+ "tokenizers": "0.23.2",
11
+ "huggingface-hub": "1.32.0",
12
+ "einops": "0.8.2"
13
+ },
14
+ "python": "3.12.10 (main, Apr 9 2025, 08:55:05) [GCC 11.4.0]",
15
+ "cuda": "12.8",
16
+ "gpu": "NVIDIA H20"
17
+ }
example_request.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "state": "The corridor ahead is blocked. The left route is clear.",
3
+ "questions": {
4
+ "move": {
5
+ "type": "choice",
6
+ "instructions": "Choose the safe route.",
7
+ "criteria": {
8
+ "left": "Take the clear route.",
9
+ "forward": "Hit the blockage."
10
+ }
11
+ },
12
+ "blocked": {
13
+ "type": "noul",
14
+ "instructions": "Is the corridor ahead blocked?"
15
+ },
16
+ "risk": {
17
+ "type": "score",
18
+ "instructions": "Risk of moving forward.",
19
+ "criteria": [
20
+ "low",
21
+ "medium",
22
+ "high"
23
+ ]
24
+ }
25
+ }
26
+ }
model_manifest.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "NeoHorse-Jev-4B",
3
+ "format": "unified multimodal backbone + independent decision pointer head; NOT chat causal LM",
4
+ "backbone_class": "Qwen3_5Model",
5
+ "backbone_dtype": "bfloat16",
6
+ "head_dtype": "float32",
7
+ "merge_dtype": "float32",
8
+ "lora_scale": 1.0,
9
+ "head_dim": 256,
10
+ "option_isolation": false,
11
+ "hybrid": true,
12
+ "temperature": 1.0,
13
+ "runtime_origin": "https://github.com/jaredpalmer/kev",
14
+ "created_unix": 1790132795.85603,
15
+ "bundle_version": "1.0.0",
16
+ "vision_source": "Qwen/Qwen3.5-4B",
17
+ "vision_tensors": 297,
18
+ "vision_finetuning": false,
19
+ "text_tensors_unchanged": true,
20
+ "pointer_head_unchanged": true,
21
+ "http_image_field": "image: inline PNG/JPEG/WebP base64 data URL"
22
+ }
package/pyproject.toml ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [build-system]
2
+ requires = ["setuptools>=69", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "neohorse-decision"
7
+ version = "1.0.0"
8
+ description = "NeoHorse-JEV prefill-only decision inference"
9
+ requires-python = ">=3.11"
10
+ dependencies = ["torch==2.8.0", "transformers==5.17.0", "safetensors==0.8.0", "pydantic==2.13.5"]
11
+
12
+ [project.optional-dependencies]
13
+ vision = ["pillow==12.3.0"]
14
+ serve = ["fastapi==0.141.1", "uvicorn==0.53.0", "starlette==1.6.0", "pillow==12.3.0"]
15
+ test = ["httpx==0.28.1"]
16
+
17
+ [project.scripts]
18
+ neohorse-decision = "neohorse_decision.cli:main"
19
+
20
+ [tool.setuptools.packages.find]
21
+ where = ["src"]
22
+
23
+ [tool.setuptools.package-data]
24
+ neohorse_decision = ["_vendor/LICENSE", "_vendor/NOTICE.md"]
package/src/neohorse_decision/__init__.py ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ from importlib.metadata import version as _package_version
2
+ from .engine import DecisionEngine
3
+
4
+ __version__ = _package_version('neohorse-decision')
5
+ __all__ = ['DecisionEngine']
package/src/neohorse_decision/_inference.py ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Offline loader for the merged NeoHorse decision model (not a chat LM)."""
2
+ import json
3
+ from pathlib import Path
4
+
5
+ import torch
6
+ from safetensors.torch import load_file
7
+ from transformers import AutoModel, AutoTokenizer
8
+ from ._vendor.model import DecisionModel, PointerHead
9
+ from ._vendor.schema import SystemOneRequest, to_record
10
+
11
+
12
+ def load_bundle(bundle, device='cuda'):
13
+ root = Path(bundle)
14
+ meta = json.loads((root / 'model_manifest.json').read_text())
15
+ torch.set_num_threads(4)
16
+ torch.backends.cuda.matmul.allow_tf32 = False
17
+ torch.backends.cudnn.allow_tf32 = False
18
+ torch.backends.cuda.enable_flash_sdp(True)
19
+ torch.backends.cuda.enable_mem_efficient_sdp(True)
20
+ tok = AutoTokenizer.from_pretrained(root / 'tokenizer', local_files_only=True)
21
+ backbone, info = AutoModel.from_pretrained(
22
+ root / 'backbone', dtype=torch.bfloat16, attn_implementation='sdpa',
23
+ local_files_only=True, output_loading_info=True)
24
+ assert not info.get('missing_keys') and not info.get('unexpected_keys'), info
25
+ assert not info.get('mismatched_keys') and not info.get('error_msgs'), info
26
+ assert type(backbone).__name__ == meta['backbone_class']
27
+ # Bypass the training constructor: it loads the original base checkpoint.
28
+ # This bundle instead supplies the complete, already merged backbone.
29
+ model = DecisionModel.__new__(DecisionModel)
30
+ torch.nn.Module.__init__(model)
31
+ # Some Transformers modules are kept FP32 during from_pretrained even when
32
+ # dtype=BF16. The evaluated serving loader explicitly cast the WHOLE merged
33
+ # backbone after merging, so reproduce that cast after reload as well.
34
+ backbone = backbone.to(device=device, dtype=torch.bfloat16)
35
+ if type(backbone).__name__ == 'Qwen3_5Model':
36
+ model.multimodal = backbone
37
+ model.lm = backbone.language_model
38
+ else:
39
+ model.lm = backbone
40
+ text_config = model.lm.config
41
+ model.head = PointerHead(text_config.hidden_size, dp=meta['head_dim'])
42
+ model.head.load_state_dict(load_file(str(root / 'pointer_head.safetensors')))
43
+ model.head = model.head.to(device=device, dtype=torch.float32)
44
+ model.pad_id = tok.pad_token_id if tok.pad_token_id is not None else 0
45
+ model.hybrid = 'linear_attention' in set(getattr(text_config, 'layer_types', None) or [])
46
+ model.option_isolation = meta['option_isolation']
47
+ model.device = device
48
+ assert model.hybrid == meta['hybrid']
49
+ assert not (model.hybrid and model.option_isolation)
50
+ model.eval()
51
+ return tok, model
52
+
53
+
54
+ def predict(tok, model, request, max_state=2048, max_branch=8192):
55
+ rec, meta = to_record(SystemOneRequest(**request))
56
+ enc = model.encode(tok, rec, strict=True, max_state=max_state, max_branch=max_branch)
57
+ with torch.inference_mode():
58
+ ps = [p.tolist() for p in model.probs(enc)]
59
+ answers = {}
60
+ for p, m in zip(ps, meta):
61
+ if m['type'] == 'choice':
62
+ answers[m['id']] = dict(type='choice', choice=m['keys'][max(range(len(p)), key=p.__getitem__)],
63
+ probabilities=dict(zip(m['keys'], p)))
64
+ elif m['type'] == 'noul':
65
+ answers[m['id']] = dict(type='noul', noul=p[1], probabilities={'false': p[0], 'true': p[1]})
66
+ else:
67
+ answers[m['id']] = dict(type='score', score=sum(i * v for i, v in enumerate(p)),
68
+ legend=m['legend'], probabilities={str(i): v for i, v in enumerate(p)})
69
+ return dict(answers=answers, input_tokens=len(enc['ids']))
70
+
package/src/neohorse_decision/_vendor/LICENSE ADDED
@@ -0,0 +1,203 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Jared Palmer
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
203
+
package/src/neohorse_decision/_vendor/NOTICE.md ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party attribution
2
+
3
+ `model.py` and `schema.py` derive from Jared Palmer's Kev (https://github.com/jaredpalmer/kev), Copyright 2026 Jared Palmer, Apache-2.0; see LICENSE.
4
+
5
+ The starting files were copied from the local runtime used for this checkpoint. Original file hashes are recorded in the accompanying model_manifest.json. The default API model identifier in `schema.py` was changed to `neohorse-jev`.
6
+
7
+ NeoHorse modifications add `NEOHORSE_SHAPE_BUCKET` as the primary environment variable, retaining `KEV_SHAPE_BUCKET` as a lower-priority compatibility alias (default 64; 1 disables MPS sequence padding). Unused date-preprocessing helpers were removed, and stale comments and the package description were refreshed. Decision encoding, probability readout, and model weights are unchanged.
8
+
9
+ The package does not fetch or import an external Kev distribution. Vendored code is versioned with this package, not automatically updated from upstream.
10
+
11
+ This notice does not assign a new license to the model weights or training data.
package/src/neohorse_decision/_vendor/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ """Bundled decision primitives; third-party attribution is recorded in NOTICE.md."""
package/src/neohorse_decision/_vendor/model.py ADDED
@@ -0,0 +1,300 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Decision model: causal LM backbone + block-causal branch mask + pointer readout."""
2
+ # Modified for NeoHorse: environment naming with a legacy alias; refreshed comments.
3
+ import math, os, re
4
+ import torch
5
+ import torch.nn as nn
6
+ import torch.nn.functional as F
7
+ from transformers import AutoModelForCausalLM, AutoTokenizer
8
+
9
+ # Reuse existing rarely-used Qwen special tokens as delimiters (state, q, opt, /opt, decide) so no
10
+ # embedding rows need to be added/trained; LoRA adapts their meaning.
11
+ SPECIAL = ["<|fim_prefix|>", "<|fim_middle|>", "<|box_start|>", "<|box_end|>", "<|fim_suffix|>"]
12
+ MAX_STATE, MAX_BRANCH = 384, 1024
13
+
14
+
15
+ def load_tokenizer(name, revision=None):
16
+ return AutoTokenizer.from_pretrained(name, revision=revision)
17
+
18
+
19
+ _SPECIAL_RE = re.compile(r"<\|([A-Za-z0-9_]+)\|>")
20
+
21
+
22
+ def user_tokens(tok, text):
23
+ """Tokenize caller-supplied text so it can never produce delimiter/control tokens (option boundaries are unforgeable).
24
+ The fast tokenizer ignores split_special_tokens, so `<|name|>` is rewritten to `<¦name¦>` before tokenizing."""
25
+ return tok(_SPECIAL_RE.sub(r"<¦\1¦>", text), add_special_tokens=False).input_ids
26
+
27
+
28
+ OPT_NONE, OPT_DECIDE = -1, -2 # values of enc["opt"]: instruction/state tokens, and the <decide> token
29
+
30
+
31
+ def encode(tok, rec, max_state=MAX_STATE, max_branch=MAX_BRANCH, strict=False, option_isolation=False):
32
+ """Pack one record: [<state> ...] then per-question [<q> instr <opt> o </opt>... <decide>].
33
+
34
+ Returns ids, seg (0 = state, k = question k), pos (branch positions restart after state),
35
+ decide_idx [Q], opt_idx [Q][K] (index of </opt> token for each option), opt (per-token option index within its
36
+ question: OPT_NONE for state/instruction, 0..K-1 for option spans, OPT_DECIDE for <decide>).
37
+
38
+ option_isolation=True: every option span is its own sub-branch (it sees state + instruction + itself only), all
39
+ option spans share the same position ids, and <decide> sits at one fixed position after the longest span. Then the
40
+ per-option representations and <decide>'s attention over them are permutation-invariant by construction.
41
+ """
42
+ state_tokens = user_tokens(tok, rec["state"])
43
+ if strict and len(state_tokens) + 1 > max_state:
44
+ raise ValueError(f"state exceeds {max_state} tokens: {len(state_tokens) + 1}")
45
+ S = [tok.convert_tokens_to_ids(SPECIAL[0])] + state_tokens[: max_state - 1]
46
+ ids, seg, pos, opt = list(S), [0] * len(S), list(range(len(S))), [OPT_NONE] * len(S)
47
+ q_id, o_id, c_id, d_id = (tok.convert_tokens_to_ids(t) for t in SPECIAL[1:])
48
+ decide_idx, opt_idx = [], []
49
+ for k, q in enumerate(rec["questions"], start=1):
50
+ instr = [q_id] + user_tokens(tok, q["instr"])
51
+ spans = [[o_id] + user_tokens(tok, o) + [c_id] for o in q["options"]]
52
+ br = instr + [t for sp in spans for t in sp] + [d_id]
53
+ if len(br) > max_branch - len(S):
54
+ raise ValueError(f"branch too long: {len(br)}")
55
+ base = len(ids); p0 = len(S)
56
+ br_opt = [OPT_NONE] * len(instr) + [j for j, sp in enumerate(spans) for _ in sp] + [OPT_DECIDE]
57
+ if option_isolation:
58
+ longest = max(len(sp) for sp in spans)
59
+ br_pos = list(range(p0, p0 + len(instr))) + [p0 + len(instr) + i for sp in spans for i in range(len(sp))] + [p0 + len(instr) + longest]
60
+ else:
61
+ br_pos = list(range(p0, p0 + len(br)))
62
+ ends, cursor = [], len(instr)
63
+ for sp in spans:
64
+ cursor += len(sp); ends.append(cursor - 1)
65
+ ids += br; seg += [k] * len(br); pos += br_pos; opt += br_opt
66
+ decide_idx.append(base + len(br) - 1); opt_idx.append([base + e for e in ends])
67
+ return {"ids": ids, "seg": seg, "pos": pos, "opt": opt, "option_isolation": option_isolation, "decide_idx": decide_idx, "opt_idx": opt_idx,
68
+ "labels": [q["label"] for q in rec["questions"]], "state_truncated": len(state_tokens) + 1 > max_state}
69
+
70
+
71
+ def branch_mask(seg, device, dtype=torch.float32):
72
+ """attend(i,j) iff j<=i and (seg[j]==0 or seg[j]==seg[i]). Returns additive [1,1,L,L]."""
73
+ return branch_mask_batch([seg], device, dtype)
74
+
75
+
76
+ def branch_mask_batch(segs, device, dtype=torch.float32, opts=None, length=None):
77
+ """Batched block-causal mask, additive [B,1,L,L], right-padded to the longest sequence.
78
+
79
+ Padded key positions are masked for every query; padded query rows keep the diagonal so no row is fully
80
+ masked (finfo.min, not -inf, so softmax stays finite either way). Real tokens never see pads because pads sit
81
+ after them (causal) and belong to no segment (-1).
82
+
83
+ opts (option isolation): within a question, an option-span token may attend to state, the instruction, and its own
84
+ span only; <decide> attends to everything in its question. Instruction tokens never see option spans (causal)."""
85
+ L = max(max(len(s) for s in segs), length or 0)
86
+ s = torch.full((len(segs), L), -1, device=device)
87
+ for b, seg in enumerate(segs):
88
+ s[b, : len(seg)] = torch.tensor(seg, device=device)
89
+ causal = torch.tril(torch.ones(L, L, dtype=torch.bool, device=device))
90
+ same = (s[:, None, :] == s[:, :, None]) | (s[:, None, :] == 0)
91
+ valid_key = (s != -1)[:, None, :]
92
+ allow = causal[None] & same & valid_key
93
+ if opts is not None:
94
+ o = torch.full((len(segs), L), OPT_NONE, device=device)
95
+ for b, op in enumerate(opts):
96
+ o[b, : len(op)] = torch.tensor(op, device=device)
97
+ key_is_option = (o[:, None, :] >= 0)
98
+ query_is_decide = (o[:, :, None] == OPT_DECIDE)
99
+ same_option = o[:, None, :] == o[:, :, None]
100
+ allow = allow & (~key_is_option | query_is_decide | same_option)
101
+ allow = allow | torch.eye(L, dtype=torch.bool, device=device)[None]
102
+ return torch.zeros(len(segs), L, L, dtype=dtype, device=device).masked_fill(~allow, torch.finfo(dtype).min)[:, None]
103
+
104
+
105
+ def rows_of(enc):
106
+ """Split a packed encoding into its state and per-question branch rows.
107
+
108
+ Returns (state_ids, state_pos, rows) with rows[k] = {"ids", "pos", "decide", "opts"}: the branch tokens of question
109
+ k with their (already state-continuing) positions, and the readout offsets *within the branch*. Feeding
110
+ state + rows[k] as one causal row is equivalent to the packed block-causal form for that question, on any
111
+ architecture: the row contains exactly the tokens question k may attend to, in the same positions."""
112
+ seg = enc["seg"]; Ls = seg.count(0)
113
+ rows, start = [], Ls
114
+ for k, (d, oi) in enumerate(zip(enc["decide_idx"], enc["opt_idx"]), start=1):
115
+ end = d + 1 # <decide> is the last token of its branch
116
+ if seg[start] != k or seg[end - 1] != k: raise ValueError("branch layout mismatch")
117
+ rows.append({"ids": enc["ids"][start:end], "pos": enc["pos"][start:end], "decide": d - start, "opts": [o - start for o in oi]})
118
+ start = end
119
+ return enc["ids"][:Ls], enc["pos"][:Ls], rows
120
+
121
+
122
+ class PointerHead(nn.Module):
123
+ def __init__(self, d, dp=256):
124
+ """dp = pointer dimension (head capacity knob)."""
125
+ super().__init__()
126
+ self.q, self.k = nn.Linear(d, dp), nn.Linear(d, dp)
127
+ self.scale = 1 / math.sqrt(dp)
128
+
129
+ def forward(self, h_decide, h_opts): # [d], [K,d] -> logits [K]
130
+ return (self.k(h_opts) @ self.q(h_decide)) * self.scale
131
+
132
+
133
+ class DecisionModel(nn.Module):
134
+ def __init__(self, name, tok, device, lora=None, revision=None, attn=None, head_dim=256, option_isolation=False, special_embeddings=False, lora_targets="all", dtype=torch.float32):
135
+ super().__init__()
136
+ # backbone only (no vocab head): we never generate text.
137
+ # eager on MPS/CPU (known-good with our float 4D mask); SDPA on CUDA (accepts arbitrary additive masks).
138
+ attn = attn or ("sdpa" if str(device).startswith("cuda") else "eager")
139
+ # dtype: fp32 for training and exact evaluation; bf16 is a serving option for large backbones (8B on a 32 GB Mac)
140
+ self.lm = AutoModelForCausalLM.from_pretrained(name, revision=revision, dtype=dtype, attn_implementation=attn).model
141
+ self.pad_id = tok.pad_token_id if tok.pad_token_id is not None else 0
142
+ # hybrid backbones (Qwen3.5: Gated DeltaNet layers, recurrent) cannot honour the block-causal mask, so every
143
+ # question runs as its own causal row continuing from the state (rows_of). Attention-only backbones keep the
144
+ # packed form; each question can attend only to its state and its own branch.
145
+ cfg = self.lm.config
146
+ self.hybrid = "linear_attention" in set(getattr(cfg, "layer_types", None) or [])
147
+ if self.hybrid and option_isolation: raise ValueError("option_isolation needs the packed mask; not available on hybrid backbones")
148
+ self.option_isolation = option_isolation
149
+ if lora:
150
+ from peft import LoraConfig, get_peft_model
151
+ extra = {"trainable_token_indices": {"embed_tokens": [tok.convert_tokens_to_ids(t) for t in SPECIAL]}} if special_embeddings else {}
152
+ targets = {"all": ["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"],
153
+ "dense": ["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"], # "all" minus the DeltaNet projections on hybrids (retention ablation)
154
+ "attn": ["q_proj", "k_proj", "v_proj", "o_proj"], "qv": ["q_proj", "v_proj"]}[lora_targets]
155
+ if self.hybrid and lora_targets in ("all", "attn"):
156
+ # Gated DeltaNet projections (transformers 5 names, verified on Qwen3_5TextModel); the mixer's out_proj too
157
+ targets = targets + ["in_proj_qkv", "in_proj_z", "in_proj_a", "in_proj_b", "out_proj"]
158
+ cfg = LoraConfig(task_type="FEATURE_EXTRACTION", r=lora, lora_alpha=2 * lora, lora_dropout=0.05, target_modules=targets, **extra)
159
+ self.lm = get_peft_model(self.lm, cfg)
160
+ self.head = PointerHead(self.lm.config.hidden_size, dp=head_dim)
161
+ self.device = device
162
+ self.to(device)
163
+
164
+ def encode(self, tok, rec, **kw):
165
+ """encode() with this model's option-isolation setting; use this from serving/eval code."""
166
+ return encode(tok, rec, option_isolation=self.option_isolation, **kw)
167
+
168
+ def hidden(self, enc):
169
+ return self.hidden_batch([enc])[0, : len(enc["ids"])]
170
+
171
+ # MPS sequence padding; 1 disables. The NeoHorse name takes precedence over the legacy alias.
172
+ SHAPE_BUCKET = int(os.environ.get("NEOHORSE_SHAPE_BUCKET", os.environ.get("KEV_SHAPE_BUCKET", "64")))
173
+
174
+ def hidden_batch(self, encs):
175
+ """[B, L_max, d] hidden states for a right-padded batch of encoded records. Pads are masked keys and sit after every
176
+ real token, so padding never changes a real token's hidden state (parity measured exact)."""
177
+ L = max(len(e["ids"]) for e in encs)
178
+ if str(self.device) == "mps" and not self.training: L = -(-L // self.SHAPE_BUCKET) * self.SHAPE_BUCKET
179
+ ids = torch.full((len(encs), L), self.pad_id, device=self.device)
180
+ pos = torch.zeros((len(encs), L), dtype=torch.long, device=self.device)
181
+ for b, e in enumerate(encs):
182
+ ids[b, : len(e["ids"])] = torch.tensor(e["ids"], device=self.device)
183
+ pos[b, : len(e["pos"])] = torch.tensor(e["pos"], device=self.device)
184
+ isolate = any(e.get("option_isolation") for e in encs)
185
+ if isolate and not all(e.get("option_isolation") for e in encs):
186
+ raise ValueError("cannot mix option-isolated and plain encodings in one batch")
187
+ lm_dtype = next(self.lm.parameters()).dtype
188
+ mask = branch_mask_batch([e["seg"] for e in encs], self.device, dtype=lm_dtype, opts=[e["opt"] for e in encs] if isolate else None, length=L)
189
+ return self.lm(input_ids=ids, position_ids=pos, attention_mask=mask).last_hidden_state.float() # head stays fp32
190
+
191
+ def _readout(self, h, enc):
192
+ return [self.head(h[d], h[torch.tensor(oi, device=self.device)]) for d, oi in zip(enc["decide_idx"], enc["opt_idx"])]
193
+
194
+ def forward_rows_batch(self, encs):
195
+ """Row form: every question of every record is one causal row = state tokens + its branch tokens, right-padded
196
+ into a single batch. Returns the same nested logits as forward_batch. Exact isolation by construction (rows are
197
+ independent); the state is recomputed per row (Q x state tokens), which training accepts; serving uses the
198
+ prefix cache instead."""
199
+ rows, owners = [], []
200
+ for b, e in enumerate(encs):
201
+ S, Sp, brs = rows_of(e)
202
+ for r in brs:
203
+ rows.append((S + r["ids"], Sp + r["pos"], len(S) + r["decide"], [len(S) + o for o in r["opts"]])); owners.append(b)
204
+ L = max(len(ids) for ids, *_ in rows)
205
+ if str(self.device) == "mps" and not self.training: L = -(-L // self.SHAPE_BUCKET) * self.SHAPE_BUCKET
206
+ ids = torch.full((len(rows), L), self.pad_id, device=self.device)
207
+ pos = torch.zeros((len(rows), L), dtype=torch.long, device=self.device)
208
+ att = torch.zeros((len(rows), L), dtype=torch.long, device=self.device)
209
+ for i, (rid, rpos, _, _) in enumerate(rows):
210
+ ids[i, : len(rid)] = torch.tensor(rid, device=self.device); pos[i, : len(rpos)] = torch.tensor(rpos, device=self.device); att[i, : len(rid)] = 1
211
+ h = self.lm(input_ids=ids, position_ids=pos, attention_mask=att).last_hidden_state.float()
212
+ out = [[] for _ in encs]
213
+ for i, (b, (_, _, d, oi)) in enumerate(zip(owners, rows)):
214
+ out[b].append(self.head(h[i, d], h[i, torch.tensor(oi, device=self.device)]))
215
+ return out
216
+
217
+ def forward(self, enc):
218
+ """Returns list of logits tensors, one per question."""
219
+ if self.hybrid: return self.forward_rows_batch([enc])[0]
220
+ return self._readout(self.hidden(enc), enc)
221
+
222
+ def forward_batch(self, encs):
223
+ """List (per record) of lists (per question) of logits, from one padded forward pass."""
224
+ if self.hybrid: return self.forward_rows_batch(encs)
225
+ hs = self.hidden_batch(encs)
226
+ return [self._readout(hs[b], e) for b, e in enumerate(encs)]
227
+
228
+ @torch.no_grad()
229
+ def probs(self, enc):
230
+ return [F.softmax(z, -1).cpu() for z in self.forward(enc)]
231
+
232
+ # --- state-prefix reuse (serving): the state is encoded once, question branches attend to its cached keys/values.
233
+ # Exact by construction: branch tokens never attend to each other across questions (block-causal mask) and the state
234
+ # never sees the branches (causal), so the state's hidden states and KV are identical with or without the branches.
235
+
236
+ def _branch_rows_from_prefix(self, enc, cache):
237
+ """Hybrid serving: replicate the cached state once per question and run the branches as causal rows (exactly the
238
+ forward_rows_batch layout, minus the recomputed state). The cache is consumed (replicated, then extended)."""
239
+ S, Sp, rows = rows_of(enc); Q = len(rows)
240
+ cache.reorder_cache(torch.zeros(Q, dtype=torch.long, device=self.device))
241
+ W = max(len(r["ids"]) for r in rows)
242
+ if str(self.device) == "mps": W = -(-W // self.SHAPE_BUCKET) * self.SHAPE_BUCKET
243
+ ids = torch.full((Q, W), self.pad_id, device=self.device); pos = torch.zeros((Q, W), dtype=torch.long, device=self.device)
244
+ att = torch.zeros((Q, len(S) + W), dtype=torch.long, device=self.device)
245
+ for i, r in enumerate(rows):
246
+ ids[i, : len(r["ids"])] = torch.tensor(r["ids"], device=self.device); pos[i, : len(r["pos"])] = torch.tensor(r["pos"], device=self.device); att[i, : len(S) + len(r["ids"])] = 1
247
+ h = self.lm(input_ids=ids, position_ids=pos, attention_mask=att, past_key_values=cache, use_cache=True).last_hidden_state.float()
248
+ return [F.softmax(self.head(h[i, r["decide"]], h[i, torch.tensor(r["opts"], device=self.device)]), -1).cpu() for i, r in enumerate(rows)]
249
+
250
+ @torch.no_grad()
251
+ def prefix(self, enc):
252
+ """Run the state tokens only. Returns (n_state_tokens, kv cache, state hidden states [Ls, d])."""
253
+ from transformers import DynamicCache
254
+ Ls = enc["seg"].count(0)
255
+ ids = torch.tensor([enc["ids"][:Ls]], device=self.device); pos = torch.tensor([enc["pos"][:Ls]], device=self.device)
256
+ # the cache must know the layer types (hybrid backbones keep recurrent + conv states per DeltaNet layer)
257
+ out = self.lm(input_ids=ids, position_ids=pos, past_key_values=DynamicCache(config=self.lm.config), use_cache=True)
258
+ return Ls, out.past_key_values, out.last_hidden_state[0].float()
259
+
260
+ @torch.no_grad()
261
+ def probs_and_prefix(self, enc):
262
+ """One full pass that also returns the state prefix (KV cropped to the state, state hidden states): a cache miss
263
+ costs a single forward pass, not two."""
264
+ from transformers import DynamicCache
265
+ Ls = enc["seg"].count(0)
266
+ if self.hybrid:
267
+ # recurrent layers cannot be cropped back to the state, so a hybrid miss is state pass + branch rows (the
268
+ # state pass is kept as the reusable prefix by running it twice? no: copy the cache before consuming it)
269
+ Ls, cache, h_state = self.prefix(enc)
270
+ import copy
271
+ return self._branch_rows_from_prefix(enc, copy.deepcopy(cache)), (Ls, cache, h_state)
272
+ ids = torch.tensor([enc["ids"]], device=self.device); pos = torch.tensor([enc["pos"]], device=self.device)
273
+ dt = next(self.lm.parameters()).dtype
274
+ mask = branch_mask_batch([enc["seg"]], self.device, dtype=dt, opts=[enc["opt"]] if enc.get("option_isolation") else None)
275
+ out = self.lm(input_ids=ids, position_ids=pos, attention_mask=mask, past_key_values=DynamicCache(config=self.lm.config), use_cache=True)
276
+ h = out.last_hidden_state[0].float()
277
+ out.past_key_values.crop(-(len(enc["ids"]) - Ls)) # keep the state only (negative = drop that many trailing tokens; positive form deprecated in transformers 5)
278
+ return [F.softmax(z, -1).cpu() for z in self._readout(h, enc)], (Ls, out.past_key_values, h[:Ls].clone())
279
+
280
+ @torch.no_grad()
281
+ def probs_with_prefix(self, enc, prefix):
282
+ """probs() for a record whose state tokens equal the cached prefix's; only the branches run. The cache is cropped
283
+ back to the state afterwards so it can be reused."""
284
+ Ls, cache, h_state = prefix
285
+ if enc["seg"].count(0) != Ls: raise ValueError("prefix does not match this record's state")
286
+ if self.hybrid:
287
+ import copy
288
+ return self._branch_rows_from_prefix(enc, copy.deepcopy(cache)) # the stored prefix stays pristine
289
+ ids = torch.tensor([enc["ids"][Ls:]], device=self.device); pos = torch.tensor([enc["pos"][Ls:]], device=self.device)
290
+ dt = next(self.lm.parameters()).dtype
291
+ mask = branch_mask_batch([enc["seg"]], self.device, dtype=dt, opts=[enc["opt"]] if enc.get("option_isolation") else None)[:, :, Ls:, :]
292
+ try:
293
+ out = self.lm(input_ids=ids, position_ids=pos, past_key_values=cache, attention_mask=mask, use_cache=True)
294
+ h = torch.cat([h_state, out.last_hidden_state[0].float()], 0)
295
+ finally:
296
+ cache.crop(-(len(enc["ids"]) - Ls))
297
+ return [F.softmax(z, -1).cpu() for z in self._readout(h, enc)]
298
+
299
+ def trainable_parameters(self):
300
+ return [p for p in self.parameters() if p.requires_grad]
package/src/neohorse_decision/_vendor/schema.py ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """TypeSafe-compatible request/response shapes (POST /v1/systemone) mapped onto the single pointer primitive.
2
+
3
+ Noul -> 2 options [false, true]; answer = p(true)
4
+ Choice -> options 'name' or 'name: desc'; answer = argmax, probabilities by name, confidence
5
+ Score -> options = ordered level descriptions; answer = expected level, legend, probabilities by index
6
+ """
7
+ # Modified for NeoHorse: default model identifier; unused date preprocessing removed.
8
+ from typing import Any, Literal, Union
9
+ from pydantic import BaseModel, Field, model_validator
10
+
11
+ JSONContent = Union[str, dict, list, int, float, bool, None]
12
+ MAX_OPTIONS = 255
13
+
14
+
15
+ class Noul(BaseModel):
16
+ type: Literal["noul"]
17
+ instructions: JSONContent
18
+ criteria: dict[str, JSONContent] | None = None
19
+
20
+
21
+ class Choice(BaseModel):
22
+ type: Literal["choice"]
23
+ instructions: JSONContent
24
+ criteria: dict[str, JSONContent]
25
+
26
+ @model_validator(mode="after")
27
+ def _check(self):
28
+ if not 1 <= len(self.criteria) <= MAX_OPTIONS: raise ValueError(f"criteria must have 1..{MAX_OPTIONS} options")
29
+ return self
30
+
31
+
32
+ class Score(BaseModel):
33
+ type: Literal["score"]
34
+ instructions: JSONContent
35
+ criteria: list[JSONContent] = Field(min_length=2, max_length=MAX_OPTIONS)
36
+
37
+
38
+ Question = Union[Noul, Choice, Score]
39
+
40
+
41
+ class SystemOneRequest(BaseModel):
42
+ state: JSONContent
43
+ model: str = "neohorse-jev"
44
+ questions: dict[str, Question] = Field(min_length=1)
45
+
46
+
47
+ def render(v: JSONContent, indent: int = 0) -> str:
48
+ """Flatten str | object | array into text the model sees. Field names are kept as labels."""
49
+ pad = " " * indent
50
+ if v is None: return ""
51
+ if isinstance(v, (str, int, float, bool)): return str(v)
52
+ if isinstance(v, list): return "\n".join(f"{pad}- {render(x, indent + 1).lstrip()}" for x in v)
53
+ return "\n".join(f"{pad}{k}:\n{render(x, indent + 1)}" if isinstance(x, (dict, list)) else f"{pad}{k}: {render(x)}" for k, x in v.items())
54
+
55
+
56
+ def option_text(name: str, desc: JSONContent) -> str:
57
+ return name if desc is None or desc == "" else f"{name}: {render(desc)}"
58
+
59
+
60
+ def to_record(req: SystemOneRequest):
61
+ """-> internal record for encode(), plus per-question metadata to map probabilities back."""
62
+ qs, meta = [], []
63
+ for qid, q in req.questions.items():
64
+ instr = render(q.instructions)
65
+ if q.type == "noul":
66
+ c = q.criteria or {}
67
+ opts = [option_text("no", c.get("false")), option_text("yes", c.get("true"))]
68
+ meta.append({"id": qid, "type": "noul"})
69
+ elif q.type == "choice":
70
+ opts = [option_text(k, v) for k, v in q.criteria.items()]
71
+ meta.append({"id": qid, "type": "choice", "keys": list(q.criteria.keys())})
72
+ else:
73
+ opts = [render(x) for x in q.criteria]
74
+ meta.append({"id": qid, "type": "score", "legend": {str(i): render(x) for i, x in enumerate(q.criteria)}})
75
+ qs.append({"instr": instr, "options": opts, "label": 0})
76
+ return {"state": render(req.state), "questions": qs}, meta
77
+
78
+
79
+ def choice_confidence(p: list[float]) -> float:
80
+ K = len(p)
81
+ return 1.0 if K == 1 else (max(p) - 1 / K) / (1 - 1 / K)
82
+
83
+
84
+ def score_confidence(p: list[float]) -> float:
85
+ """Approximation of TypeSafe's 'distance from the modal level' statistic (exact formula unpublished):
86
+ 1 - E|level - mode| / (L - 1)."""
87
+ L = len(p); mode = max(range(L), key=lambda i: p[i])
88
+ return 1.0 - sum(pi * abs(i - mode) for i, pi in enumerate(p)) / (L - 1)
89
+
90
+
91
+ def r2(x: float) -> float:
92
+ return round(float(x), 2)
93
+
94
+
95
+ def to_answers(probs: list[list[float]], meta: list[dict]) -> dict[str, Any]:
96
+ out = {}
97
+ for p, m in zip(probs, meta):
98
+ if m["type"] == "noul":
99
+ out[m["id"]] = {"type": "noul", "noul": r2(p[1])}
100
+ elif m["type"] == "choice":
101
+ dist = {k: r2(v) for k, v in zip(m["keys"], p)}
102
+ out[m["id"]] = {"type": "choice", "choice": m["keys"][max(range(len(p)), key=lambda i: p[i])], "confidence": r2(choice_confidence(p)), "probabilities": dist}
103
+ else:
104
+ score = sum(i * pi for i, pi in enumerate(p))
105
+ out[m["id"]] = {"type": "score", "score": r2(score), "legend": m["legend"], "probabilities": {str(i): r2(v) for i, v in enumerate(p)}, "confidence": r2(score_confidence(p))}
106
+ return out
107
+
108
+
109
+ def output_tokens(tok, answers: dict) -> int:
110
+ """Billing-style figure: tokens of the serialised answers. Not a measure of generation (there is none)."""
111
+ import json
112
+ return len(tok(json.dumps(answers), add_special_tokens=False).input_ids)
package/src/neohorse_decision/cli.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import argparse
2
+ import json
3
+ import os
4
+ from pathlib import Path
5
+ from .engine import DecisionEngine
6
+
7
+
8
+ def main():
9
+ p = argparse.ArgumentParser()
10
+ p.add_argument('command', choices=['predict', 'serve'])
11
+ p.add_argument('--model-dir', required=True)
12
+ p.add_argument('--device', default='cuda')
13
+ p.add_argument('--request')
14
+ p.add_argument('--host', default='127.0.0.1')
15
+ p.add_argument('--port', default=8080, type=int)
16
+ a = p.parse_args()
17
+ if a.command == 'predict' and not a.request:
18
+ p.error('predict requires --request')
19
+ engine = DecisionEngine(a.model_dir, a.device)
20
+ if a.command == 'predict':
21
+ print(json.dumps(engine.predict(json.loads(Path(a.request).read_text())), ensure_ascii=False, indent=2))
22
+ else:
23
+ import uvicorn
24
+ from .server import create_app
25
+ uvicorn.run(create_app(engine, os.environ.get('NEOHORSE_API_KEY')), host=a.host, port=a.port, workers=1)
package/src/neohorse_decision/engine.py ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import json
2
+ import threading
3
+ from ._inference import load_bundle, predict
4
+ from ._vendor.schema import SystemOneRequest, to_record
5
+ from .systemone import MODEL_ID, MODEL_NAMES
6
+
7
+
8
+ class BusyError(RuntimeError):
9
+ pass
10
+
11
+
12
+ class DecisionEngine:
13
+ """One serialized GPU worker. Requests are never silently truncated."""
14
+ model_id = MODEL_ID
15
+
16
+ def __init__(self, model_dir, device='cuda', max_state=2048, max_branch=8192,
17
+ max_questions=16, max_tokens=32768):
18
+ self.model_dir, self.device = model_dir, device
19
+ self.tokenizer, self.model = load_bundle(model_dir, device)
20
+ self.max_state, self.max_branch = max_state, max_branch
21
+ self.max_questions, self.max_tokens = max_questions, max_tokens
22
+ self._lock = threading.Lock()
23
+
24
+ def predict(self, request):
25
+ if len(json.dumps(request, ensure_ascii=False).encode()) > 1024 * 1024:
26
+ raise ValueError('Request exceeds 1 MiB')
27
+ req = SystemOneRequest(**request)
28
+ if req.model not in MODEL_NAMES:
29
+ raise ValueError('Unknown model; use ' + self.model_id)
30
+ if len(req.questions) > self.max_questions:
31
+ raise ValueError('Too many questions')
32
+ if not self._lock.acquire(blocking=False):
33
+ raise BusyError('Model busy; retry later')
34
+ try:
35
+ rec, _ = to_record(req)
36
+ enc = self.model.encode(self.tokenizer, rec, strict=True,
37
+ max_state=self.max_state, max_branch=self.max_branch)
38
+ # Hybrid execution repeats the state in each independent question row.
39
+ state_length = sum(s == 0 for s in enc['seg'])
40
+ effective_tokens = len(enc['ids']) + (len(req.questions) - 1) * state_length
41
+ if effective_tokens > self.max_tokens:
42
+ raise ValueError('Request exceeds total expanded token budget')
43
+ result = predict(self.tokenizer, self.model, request, self.max_state, self.max_branch)
44
+ return dict(model=self.model_id, **result)
45
+ finally:
46
+ self._lock.release()
package/src/neohorse_decision/image_input.py ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Bounded inline image decoding. Never fetch URLs or open caller-supplied paths."""
2
+ import base64
3
+ import binascii
4
+ from io import BytesIO
5
+ import warnings
6
+
7
+ MAX_IMAGE_BYTES = 4 * 1024 * 1024
8
+ MAX_PIXELS = 4 * 1024 * 1024
9
+ MIME_FORMAT = {'image/png': 'PNG', 'image/jpeg': 'JPEG', 'image/webp': 'WEBP'}
10
+
11
+ def decode_image(value):
12
+ from PIL import Image, UnidentifiedImageError
13
+ if not isinstance(value, str) or not value.startswith('data:') or ',' not in value:
14
+ raise ValueError('image must be a data:image/png|jpeg|webp;base64,... string; URLs and paths are not supported')
15
+ header, encoded = value.split(',', 1)
16
+ mime = header[5:].removesuffix(';base64')
17
+ if header != 'data:' + mime + ';base64' or mime not in MIME_FORMAT:
18
+ raise ValueError('image supports only base64 PNG, JPEG or WebP data URLs')
19
+ if len(encoded) > 4 * ((MAX_IMAGE_BYTES + 2) // 3):
20
+ raise ValueError('Encoded image exceeds 4 MiB decoded-byte limit')
21
+ try:
22
+ data = base64.b64decode(encoded, validate=True)
23
+ except (binascii.Error, ValueError) as exc:
24
+ raise ValueError('Invalid image base64') from exc
25
+ if not data or len(data) > MAX_IMAGE_BYTES:
26
+ raise ValueError('Image must contain 1 byte to 4 MiB')
27
+ try:
28
+ with warnings.catch_warnings():
29
+ warnings.simplefilter('error', Image.DecompressionBombWarning)
30
+ with Image.open(BytesIO(data)) as source:
31
+ if source.format != MIME_FORMAT[mime]:
32
+ raise ValueError('Image MIME type does not match decoded format')
33
+ if source.width * source.height > MAX_PIXELS:
34
+ raise ValueError('Image exceeds 4 megapixels')
35
+ if getattr(source, 'n_frames', 1) != 1:
36
+ raise ValueError('Animated/multi-frame images are not supported')
37
+ source.load()
38
+ return source.convert('RGB')
39
+ except (UnidentifiedImageError, OSError, Image.DecompressionBombError,
40
+ Image.DecompressionBombWarning) as exc:
41
+ raise ValueError('Invalid, truncated or oversized image') from exc
package/src/neohorse_decision/server.py ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import hmac
2
+ import json
3
+ import threading
4
+ from fastapi import FastAPI, HTTPException, Request
5
+ from fastapi.responses import JSONResponse
6
+ from pydantic import ValidationError
7
+ from starlette.concurrency import run_in_threadpool
8
+ from . import __version__
9
+ from .engine import BusyError
10
+ from .systemone import validate_request, format_response, with_confidence
11
+ from .image_input import decode_image
12
+
13
+
14
+ def create_app(engine, api_key=None, *, vision_engine=None):
15
+ app = FastAPI(title='NeoHorse Decision', version=__version__)
16
+ if vision_engine is None and hasattr(getattr(engine, 'model', None), 'multimodal'):
17
+ from .vision import VisionDecisionEngine
18
+ vision_engine = VisionDecisionEngine(engine.model_dir, engine.device, text_engine=engine)
19
+ image_lock = threading.Lock()
20
+
21
+ def image_predict(value, encoded):
22
+ if not image_lock.acquire(blocking=False):
23
+ raise BusyError('Image worker busy')
24
+ try:
25
+ image = decode_image(encoded)
26
+ try:
27
+ return vision_engine.predict(value, image)
28
+ finally:
29
+ image.close()
30
+ finally:
31
+ image_lock.release()
32
+
33
+ @app.get('/health')
34
+ def health():
35
+ return {'status': 'ready', 'model': engine.model_id,
36
+ 'input_modalities': ['text', 'image'] if vision_engine is not None else ['text']}
37
+
38
+ async def handle(request: Request, systemone=False):
39
+ if api_key and not hmac.compare_digest(request.headers.get('authorization', ''), 'Bearer ' + api_key):
40
+ raise HTTPException(401, 'Invalid bearer token')
41
+ body = bytearray()
42
+ async for chunk in request.stream():
43
+ body.extend(chunk)
44
+ if len(body) > 8 * 1024 * 1024:
45
+ raise HTTPException(413, 'Request exceeds 8 MiB')
46
+ try:
47
+ try:
48
+ value = json.loads(body)
49
+ except (ValueError, UnicodeError, RecursionError):
50
+ if len(body) > 1024 * 1024:
51
+ raise HTTPException(413, 'Text request exceeds 1 MiB')
52
+ raise ValueError('Invalid JSON')
53
+ if not isinstance(value, dict):
54
+ raise ValueError('Request must be an object')
55
+ if any(k in value for k in ('images', 'image_url', 'image_base64', 'video')):
56
+ raise ValueError('Use the single top-level image data URL field')
57
+ has_image = 'image' in value
58
+ encoded = value.pop('image', None)
59
+ if (not has_image and len(body) > 1024 * 1024) or len(json.dumps(value, ensure_ascii=False).encode()) > 1024 * 1024:
60
+ raise HTTPException(413, 'Text request fields exceed 1 MiB')
61
+ if has_image:
62
+ if vision_engine is None:
63
+ raise ValueError('This model has no enabled vision adapter')
64
+ if not isinstance(value.get('questions'), dict) or len(value['questions']) != 1:
65
+ raise ValueError('Image requests require exactly one question')
66
+ if systemone:
67
+ value = validate_request(value)
68
+ if has_image:
69
+ result = await run_in_threadpool(image_predict, value, encoded)
70
+ else:
71
+ result = await run_in_threadpool(engine.predict, value)
72
+ if systemone:
73
+ result = format_response(result, engine.tokenizer)
74
+ return JSONResponse(result, headers={
75
+ 'X-NeoHorse-Confidence': 'local-distribution-statistic-v1',
76
+ 'X-NeoHorse-Usage': 'local-tokenizer-not-jev-billing'})
77
+ return JSONResponse(with_confidence(result), headers={
78
+ 'X-NeoHorse-Confidence': 'local-distribution-statistic-v1'})
79
+ except BusyError:
80
+ return JSONResponse({'detail': 'Model busy; retry later'}, status_code=529 if systemone else 429, headers={'Retry-After': '1'})
81
+ except (ValueError, TypeError, ValidationError) as e:
82
+ raise HTTPException(422, str(e)) from e
83
+ @app.post('/v1/decision')
84
+ async def decision(request: Request):
85
+ return await handle(request)
86
+
87
+ @app.post('/v1/systemone')
88
+ async def systemone(request: Request):
89
+ return await handle(request, systemone=True)
90
+ return app
package/src/neohorse_decision/systemone.py ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """System One wire adapter; not official Jev calibration or billing."""
2
+ import json
3
+ from ._vendor.schema import SystemOneRequest, choice_confidence, score_confidence
4
+
5
+ MODEL_ID = 'NeoHorse-Jev-4B'
6
+ MODEL_NAMES = {MODEL_ID, 'TokenRhythm/' + MODEL_ID, 'neohorse-jev',
7
+ 'NeoHorse-JEV-4B', 'TokenRhythm/NeoHorse-JEV-4B'}
8
+
9
+ def validate_request(value):
10
+ if not isinstance(value.get('model'), str) or value['model'] not in MODEL_NAMES:
11
+ raise ValueError('model is required; use NeoHorse-Jev-4B (not a Jev model alias)')
12
+ if not isinstance(value.get('state'), (str, dict, list)):
13
+ raise ValueError('state must be string, object or array')
14
+ req = SystemOneRequest(**{**value, 'model': 'neohorse-jev'})
15
+ def description(v):
16
+ return v is None or isinstance(v, (str, dict, list))
17
+ for q in req.questions.values():
18
+ if not description(q.instructions):
19
+ raise ValueError('instructions must be string, object, array or null')
20
+ if q.type == 'score':
21
+ if not 2 <= len(q.criteria) <= 10:
22
+ raise ValueError('System One Score requires 2..10 levels')
23
+ values = q.criteria
24
+ elif q.type == 'noul':
25
+ if q.criteria and not set(q.criteria) <= {'true', 'false'}:
26
+ raise ValueError('Noul criteria keys must be true or false')
27
+ values = (q.criteria or {}).values()
28
+ else:
29
+ values = q.criteria.values()
30
+ if any(not description(v) for v in values):
31
+ raise ValueError('criteria descriptions must be string, object, array or null')
32
+ return req.model_dump()
33
+
34
+ def with_confidence(result):
35
+ """Add the same local confidence statistic without removing native fields."""
36
+ answers = {}
37
+ for name, answer in result['answers'].items():
38
+ a = dict(answer)
39
+ if a['type'] in ('choice', 'score'):
40
+ p = list(a['probabilities'].values())
41
+ confidence = choice_confidence(p) if a['type'] == 'choice' else score_confidence(p)
42
+ a['confidence'] = min(1.0, max(0.0, confidence))
43
+ answers[name] = a
44
+ return {**result, 'answers': answers}
45
+
46
+ def format_response(result, tokenizer):
47
+ answers = with_confidence(result)['answers']
48
+ for name, a in answers.items():
49
+ if a['type'] == 'noul':
50
+ answers[name] = {'type': 'noul', 'noul': a['noul']}
51
+ output_tokens = len(tokenizer(json.dumps(answers, ensure_ascii=False, separators=(',', ':')),
52
+ add_special_tokens=False).input_ids)
53
+ usage = {'input_tokens': result['input_tokens'], 'output_tokens': output_tokens}
54
+ if 'image_tokens' in result:
55
+ usage['image_tokens'] = result['image_tokens']
56
+ return {'model': MODEL_ID, 'answers': answers, 'usage': usage}
package/src/neohorse_decision/vision.py ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Local image decisions using the unified NeoHorse multimodal backbone.
2
+
3
+ Local PIL image inference; HTTP transport is implemented in image_input.py/server.py.
4
+ """
5
+ import json
6
+ from pathlib import Path
7
+ import torch
8
+ from neohorse_decision import DecisionEngine
9
+ from neohorse_decision.engine import BusyError
10
+ from neohorse_decision._vendor.schema import SystemOneRequest, to_record
11
+ from neohorse_decision.systemone import MODEL_ID, MODEL_NAMES, with_confidence
12
+
13
+
14
+ class VisionDecisionEngine:
15
+ """One local image and one Choice/Noul/Score question per visual request."""
16
+
17
+ def __init__(self, model_dir, device='cuda', *, text_engine=None):
18
+ from transformers.models.qwen2_vl.image_processing_pil_qwen2_vl import Qwen2VLImageProcessorPil
19
+ self.root=Path(model_dir)
20
+ self.text=text_engine if text_engine is not None else DecisionEngine(self.root,device=device)
21
+ self.device=device
22
+ if not hasattr(self.text.model,'multimodal'):
23
+ raise ValueError('VisionDecisionEngine requires the unified multimodal backbone')
24
+ self.wrapper=self.text.model.multimodal
25
+ config=self.wrapper.config
26
+ self.visual_tensor_count=len(self.wrapper.visual.state_dict())
27
+ self.processor=Qwen2VLImageProcessorPil.from_pretrained(self.root/'backbone',local_files_only=True,
28
+ size={'shortest_edge':65536,'longest_edge':1048576})
29
+ for token,expected in (('<|image_pad|>',config.image_token_id),('<|vision_start|>',config.vision_start_token_id),('<|vision_end|>',config.vision_end_token_id)):
30
+ if self.text.tokenizer.convert_tokens_to_ids(token)!=expected:
31
+ raise ValueError('Vision token ID mismatch: '+token)
32
+
33
+ def predict(self,request,image=None,*,through_wrapper=False):
34
+ if image is None and not through_wrapper:
35
+ return with_confidence(self.text.predict(request))
36
+ req=SystemOneRequest(**request)
37
+ if req.model not in MODEL_NAMES:raise ValueError('Unknown model; use '+MODEL_ID)
38
+ if len(req.questions)!=1:raise ValueError('Visual adapter currently supports one question per request')
39
+ if len(json.dumps(request,ensure_ascii=False).encode())>1024*1024:raise ValueError('Request exceeds 1 MiB')
40
+ if image is not None:
41
+ from PIL import Image
42
+ if not isinstance(image,Image.Image):raise TypeError('image must be a locally decoded PIL image')
43
+ if image.width*image.height>4194304:raise ValueError('Source image exceeds 4 megapixels')
44
+ if not self.text._lock.acquire(blocking=False):raise BusyError('Model busy; retry later')
45
+ try:
46
+ rec,metadata=to_record(req)
47
+ enc=self.text.model.encode(self.text.tokenizer,rec,strict=True,max_state=self.text.max_state,max_branch=self.text.max_branch)
48
+ prefix=[];extra={};image_tokens=0
49
+ if image is not None:
50
+ batch=self.processor(images=image.convert('RGB'),return_tensors='pt')
51
+ image_tokens=int(batch['image_grid_thw'][0].prod())//self.processor.merge_size**2
52
+ if not 0<image_tokens<=1024:raise ValueError('Processed image exceeds 1024 visual tokens')
53
+ prefix=[self.wrapper.config.vision_start_token_id]+[self.wrapper.config.image_token_id]*image_tokens+[self.wrapper.config.vision_end_token_id]
54
+ extra={k:batch[k].to(self.device) for k in ('pixel_values','image_grid_thw')}
55
+ if len(enc['ids'])+len(prefix)>12288:raise ValueError('Expanded multimodal input exceeds 12288 tokens')
56
+ ids=torch.tensor([[enc['ids'][0]]+prefix+enc['ids'][1:]],device=self.device)
57
+ # Do not carry positional state across independent images/requests.
58
+ self.wrapper.rope_deltas=None
59
+ with torch.inference_mode():
60
+ h=self.wrapper(input_ids=ids,attention_mask=torch.ones_like(ids),
61
+ mm_token_type_ids=(ids==self.wrapper.config.image_token_id).long(),use_cache=False,**extra).last_hidden_state[0].float()
62
+ delta=len(prefix)
63
+ z=self.text.model.head(h[enc['decide_idx'][0]+delta],h[torch.tensor(enc['opt_idx'][0],device=self.device)+delta])
64
+ p=z.softmax(-1).cpu().tolist()
65
+ m=metadata[0]
66
+ if not all(torch.isfinite(torch.tensor(p))):raise RuntimeError('Non-finite output probabilities')
67
+ if m['type']=='choice':
68
+ answer=dict(type='choice',choice=m['keys'][max(range(len(p)),key=p.__getitem__)],probabilities=dict(zip(m['keys'],p)))
69
+ elif m['type']=='noul':
70
+ answer=dict(type='noul',noul=p[1],probabilities={'false':p[0],'true':p[1]})
71
+ else:
72
+ answer=dict(type='score',score=sum(i*v for i,v in enumerate(p)),legend=m['legend'],probabilities={str(i):v for i,v in enumerate(p)})
73
+ return with_confidence(dict(model=MODEL_ID,answers={m['id']:answer},input_tokens=len(enc['ids'])+delta,image_tokens=image_tokens))
74
+ finally:
75
+ self.text._lock.release()
pointer_head.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:467ae48977b5c4bf87dd1db29021199e0c3401c52c5b57a1eccde679c0bade26
3
+ size 5245232
tokenizer/chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
tokenizer/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer/tokenizer_config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": true,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "unk_token": null,
29
+ "video_token": "<|video_pad|>",
30
+ "vision_bos_token": "<|vision_start|>",
31
+ "vision_eos_token": "<|vision_end|>"
32
+ }
vision/LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
vision/README.md ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Image Inference Examples
2
+
3
+ See the [Deployment guide](https://huggingface.co/TokenRhythm/NeoHorse-Jev-4B/blob/main/DEPLOYMENT.md) for environment setup, image request formats, response fields, limits, and error handling.
4
+
5
+ | File | Purpose |
6
+ | --- | --- |
7
+ | `example.py` | Run image-based decisions locally |
8
+ | `example_request.json` | Sample question; replace it with your own single-question request |
9
+ | `http_example.py` | Encode a local image and send it to the HTTP API |
10
+ | `predictor.py` | Import wrapper for the installed `neohorse_decision.vision.VisionDecisionEngine` |
11
+ | `base_vision_provenance.json` | Record the source of the vision weights |
12
+ | `verification.json` | Record basic vision smoke tests and consistency checks |
13
+ | `LICENSE` | License for the vision components |
14
+
15
+ The language and vision weights are stored together in `backbone/`. The decision head is stored in `pointer_head.safetensors` at the model root. The vision weights come from Qwen3.5-4B and have not undergone additional vision training. The hashes of separate vision files in the provenance record are for traceability only; those files do not need to be loaded separately at runtime.
16
+
17
+ After installing the matching runtime package, set `MODEL_DIR` to the complete downloaded model directory. You can then run these commands from any working directory:
18
+
19
+ ```bash
20
+ CUDA_VISIBLE_DEVICES=0 python "$MODEL_DIR/vision/example.py" \
21
+ --model-dir "$MODEL_DIR" --image /path/to/image.png \
22
+ --request "$MODEL_DIR/vision/example_request.json"
23
+
24
+ python "$MODEL_DIR/vision/http_example.py" --image /path/to/image.png \
25
+ --base-url http://127.0.0.1:8080 --endpoint systemone
26
+ ```
27
+
28
+ Each image request supports one static image and one question, using Noul, Choice, or Score. The basic vision verification records are not a general visual capability evaluation.
vision/__init__.py ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ """Optional local-image adapter shipped alongside the model, not the HTTP runtime."""
2
+ from .predictor import VisionDecisionEngine
vision/base_vision_provenance.json ADDED
@@ -0,0 +1,2795 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "vision-v1",
3
+ "base_model": "Qwen/Qwen3.5-4B",
4
+ "source_index_sha256": "cf3f798ee02ba45f9622aa8892a47369ab667d0afbf154ee7c2212de42e6302d",
5
+ "copied_metadata": {
6
+ "config.json": "ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670",
7
+ "preprocessor_config.json": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516",
8
+ "LICENSE": "50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32"
9
+ },
10
+ "tensor_count": 297,
11
+ "parameter_count": 333514240,
12
+ "artifact_sha256": "86a8e4f2a5373a3d1f2151216b549791528ba6e0aa87da778e99c20392b74f98",
13
+ "tensors": [
14
+ {
15
+ "source_key": "model.visual.blocks.0.attn.proj.bias",
16
+ "stored_key": "blocks.0.attn.proj.bias",
17
+ "shape": [
18
+ 1024
19
+ ],
20
+ "dtype": "torch.bfloat16",
21
+ "sha256": "35e988e89cb4c9d2de166a7243a57da738cede31bdc7ffea79db195ab87f9c71"
22
+ },
23
+ {
24
+ "source_key": "model.visual.blocks.0.attn.proj.weight",
25
+ "stored_key": "blocks.0.attn.proj.weight",
26
+ "shape": [
27
+ 1024,
28
+ 1024
29
+ ],
30
+ "dtype": "torch.bfloat16",
31
+ "sha256": "78cca0b25e1e134deaf8dfc8c64a3fd06bde4cac1a29ebbf53657aca8a60e78b"
32
+ },
33
+ {
34
+ "source_key": "model.visual.blocks.0.attn.qkv.bias",
35
+ "stored_key": "blocks.0.attn.qkv.bias",
36
+ "shape": [
37
+ 3072
38
+ ],
39
+ "dtype": "torch.bfloat16",
40
+ "sha256": "7daad53c4533bbc09381760e5066eeee505137254ff14b6e16c2b7f9936958a9"
41
+ },
42
+ {
43
+ "source_key": "model.visual.blocks.0.attn.qkv.weight",
44
+ "stored_key": "blocks.0.attn.qkv.weight",
45
+ "shape": [
46
+ 3072,
47
+ 1024
48
+ ],
49
+ "dtype": "torch.bfloat16",
50
+ "sha256": "8590f79f0af5e1b8b2e69baa28795df9986bc4a6f338eb06a07ad51cf8a796e7"
51
+ },
52
+ {
53
+ "source_key": "model.visual.blocks.0.mlp.linear_fc1.bias",
54
+ "stored_key": "blocks.0.mlp.linear_fc1.bias",
55
+ "shape": [
56
+ 4096
57
+ ],
58
+ "dtype": "torch.bfloat16",
59
+ "sha256": "49b76a4ae4cf5e8c9ca03ab2125655e9697e48178752d00781e0e2880a4ca04b"
60
+ },
61
+ {
62
+ "source_key": "model.visual.blocks.0.mlp.linear_fc1.weight",
63
+ "stored_key": "blocks.0.mlp.linear_fc1.weight",
64
+ "shape": [
65
+ 4096,
66
+ 1024
67
+ ],
68
+ "dtype": "torch.bfloat16",
69
+ "sha256": "9e81579c6093752a9b710c68bb995fc2f2ca51a67068a363621781f9132c20be"
70
+ },
71
+ {
72
+ "source_key": "model.visual.blocks.0.mlp.linear_fc2.bias",
73
+ "stored_key": "blocks.0.mlp.linear_fc2.bias",
74
+ "shape": [
75
+ 1024
76
+ ],
77
+ "dtype": "torch.bfloat16",
78
+ "sha256": "39011ac5434dc5db9edab8ab64e7cc2b6dde8f5ee4c43bbfa8e4a9c9b1c03a8c"
79
+ },
80
+ {
81
+ "source_key": "model.visual.blocks.0.mlp.linear_fc2.weight",
82
+ "stored_key": "blocks.0.mlp.linear_fc2.weight",
83
+ "shape": [
84
+ 1024,
85
+ 4096
86
+ ],
87
+ "dtype": "torch.bfloat16",
88
+ "sha256": "05a82c58e7d798848736214ac4b4df9527e8e3f8d2f6502ac91b423c072a7436"
89
+ },
90
+ {
91
+ "source_key": "model.visual.blocks.0.norm1.bias",
92
+ "stored_key": "blocks.0.norm1.bias",
93
+ "shape": [
94
+ 1024
95
+ ],
96
+ "dtype": "torch.bfloat16",
97
+ "sha256": "6c7c81c3d295c17e4496fc40009859df63af957b498c30cad9c214b91456bf57"
98
+ },
99
+ {
100
+ "source_key": "model.visual.blocks.0.norm1.weight",
101
+ "stored_key": "blocks.0.norm1.weight",
102
+ "shape": [
103
+ 1024
104
+ ],
105
+ "dtype": "torch.bfloat16",
106
+ "sha256": "0928383981c62db2ce3ffd152cf9f64b84d1dd195546ef7d90bdfef39fbf1caa"
107
+ },
108
+ {
109
+ "source_key": "model.visual.blocks.0.norm2.bias",
110
+ "stored_key": "blocks.0.norm2.bias",
111
+ "shape": [
112
+ 1024
113
+ ],
114
+ "dtype": "torch.bfloat16",
115
+ "sha256": "c18e1278761361763c4cc471083ba75df6bf6cd96e21f2ba4c48dc092546c9a7"
116
+ },
117
+ {
118
+ "source_key": "model.visual.blocks.0.norm2.weight",
119
+ "stored_key": "blocks.0.norm2.weight",
120
+ "shape": [
121
+ 1024
122
+ ],
123
+ "dtype": "torch.bfloat16",
124
+ "sha256": "8a238107a8b61536f2e5b98d7846f8aaebac2919133175fcbf0369fb80174998"
125
+ },
126
+ {
127
+ "source_key": "model.visual.blocks.1.attn.proj.bias",
128
+ "stored_key": "blocks.1.attn.proj.bias",
129
+ "shape": [
130
+ 1024
131
+ ],
132
+ "dtype": "torch.bfloat16",
133
+ "sha256": "8299dee2cfc5fde55fe5344ed362a850b1fba1f92a76ba0972e2ceb678cfb3a6"
134
+ },
135
+ {
136
+ "source_key": "model.visual.blocks.1.attn.proj.weight",
137
+ "stored_key": "blocks.1.attn.proj.weight",
138
+ "shape": [
139
+ 1024,
140
+ 1024
141
+ ],
142
+ "dtype": "torch.bfloat16",
143
+ "sha256": "5bb8eee8d943ab25d9bc89e807d5a78e7437be7f34492f4b41169cdba254cde1"
144
+ },
145
+ {
146
+ "source_key": "model.visual.blocks.1.attn.qkv.bias",
147
+ "stored_key": "blocks.1.attn.qkv.bias",
148
+ "shape": [
149
+ 3072
150
+ ],
151
+ "dtype": "torch.bfloat16",
152
+ "sha256": "9a266b396e7792f6869dfc1dd21ea863926aa62cceb79219e146dba35b5436cd"
153
+ },
154
+ {
155
+ "source_key": "model.visual.blocks.1.attn.qkv.weight",
156
+ "stored_key": "blocks.1.attn.qkv.weight",
157
+ "shape": [
158
+ 3072,
159
+ 1024
160
+ ],
161
+ "dtype": "torch.bfloat16",
162
+ "sha256": "f0462fd9bd00f5690f047328af2268dd7640802dac22c46a1e669581358eb229"
163
+ },
164
+ {
165
+ "source_key": "model.visual.blocks.1.mlp.linear_fc1.bias",
166
+ "stored_key": "blocks.1.mlp.linear_fc1.bias",
167
+ "shape": [
168
+ 4096
169
+ ],
170
+ "dtype": "torch.bfloat16",
171
+ "sha256": "7cfa4b2269ec4d386a42dda3edff2da7aad8286b0bbfbaa213c83cd5c8b09c89"
172
+ },
173
+ {
174
+ "source_key": "model.visual.blocks.1.mlp.linear_fc1.weight",
175
+ "stored_key": "blocks.1.mlp.linear_fc1.weight",
176
+ "shape": [
177
+ 4096,
178
+ 1024
179
+ ],
180
+ "dtype": "torch.bfloat16",
181
+ "sha256": "56c7abf157f9a58f5c05ce6bbbf6720ce96aee15818f18cb479412146fc1e5ba"
182
+ },
183
+ {
184
+ "source_key": "model.visual.blocks.1.mlp.linear_fc2.bias",
185
+ "stored_key": "blocks.1.mlp.linear_fc2.bias",
186
+ "shape": [
187
+ 1024
188
+ ],
189
+ "dtype": "torch.bfloat16",
190
+ "sha256": "65162e4fd05745ccd3466431d26ccd304459dfbcb5c97633025bf3c3a9077d86"
191
+ },
192
+ {
193
+ "source_key": "model.visual.blocks.1.mlp.linear_fc2.weight",
194
+ "stored_key": "blocks.1.mlp.linear_fc2.weight",
195
+ "shape": [
196
+ 1024,
197
+ 4096
198
+ ],
199
+ "dtype": "torch.bfloat16",
200
+ "sha256": "354b3c467f3fdd8efe7a2326c7dc553cea0a6f2479e0f834741105649c6abe83"
201
+ },
202
+ {
203
+ "source_key": "model.visual.blocks.1.norm1.bias",
204
+ "stored_key": "blocks.1.norm1.bias",
205
+ "shape": [
206
+ 1024
207
+ ],
208
+ "dtype": "torch.bfloat16",
209
+ "sha256": "2d04662565c9020bd762a273e9aced97a3cd47e4361f3ad4a4c25ff5db2423af"
210
+ },
211
+ {
212
+ "source_key": "model.visual.blocks.1.norm1.weight",
213
+ "stored_key": "blocks.1.norm1.weight",
214
+ "shape": [
215
+ 1024
216
+ ],
217
+ "dtype": "torch.bfloat16",
218
+ "sha256": "cc7d59e07d52aa2bb7300a0d6e5046026ed7ac45989d6831a14fd9b1934eb537"
219
+ },
220
+ {
221
+ "source_key": "model.visual.blocks.1.norm2.bias",
222
+ "stored_key": "blocks.1.norm2.bias",
223
+ "shape": [
224
+ 1024
225
+ ],
226
+ "dtype": "torch.bfloat16",
227
+ "sha256": "70ff4e7bcf3cec5152c5a60d648a426d1a88c94bb1b643109a50cacf8d0c0d28"
228
+ },
229
+ {
230
+ "source_key": "model.visual.blocks.1.norm2.weight",
231
+ "stored_key": "blocks.1.norm2.weight",
232
+ "shape": [
233
+ 1024
234
+ ],
235
+ "dtype": "torch.bfloat16",
236
+ "sha256": "57141cf054485c6b5c69e9b29ea4bda9131d0dad6ad04d071a550d7453563138"
237
+ },
238
+ {
239
+ "source_key": "model.visual.blocks.10.attn.proj.bias",
240
+ "stored_key": "blocks.10.attn.proj.bias",
241
+ "shape": [
242
+ 1024
243
+ ],
244
+ "dtype": "torch.bfloat16",
245
+ "sha256": "b6ba11d898bf0c80afcfa4e14841d624b80963a76c5591c936f7cd6f908ad446"
246
+ },
247
+ {
248
+ "source_key": "model.visual.blocks.10.attn.proj.weight",
249
+ "stored_key": "blocks.10.attn.proj.weight",
250
+ "shape": [
251
+ 1024,
252
+ 1024
253
+ ],
254
+ "dtype": "torch.bfloat16",
255
+ "sha256": "59da85b56f1d458e6f1c999943758a4d1bb822cfeb8f0f1bbd57f9a06fbfa793"
256
+ },
257
+ {
258
+ "source_key": "model.visual.blocks.10.attn.qkv.bias",
259
+ "stored_key": "blocks.10.attn.qkv.bias",
260
+ "shape": [
261
+ 3072
262
+ ],
263
+ "dtype": "torch.bfloat16",
264
+ "sha256": "0c02d5b3cc37a0a36359afb7dbdb76e1aa030428bd9975364e7d9f5cb5c6652d"
265
+ },
266
+ {
267
+ "source_key": "model.visual.blocks.10.attn.qkv.weight",
268
+ "stored_key": "blocks.10.attn.qkv.weight",
269
+ "shape": [
270
+ 3072,
271
+ 1024
272
+ ],
273
+ "dtype": "torch.bfloat16",
274
+ "sha256": "25860be82650a047b70e6a6da621ffd88cd66a1311d1d96cedf762c9571098a5"
275
+ },
276
+ {
277
+ "source_key": "model.visual.blocks.10.mlp.linear_fc1.bias",
278
+ "stored_key": "blocks.10.mlp.linear_fc1.bias",
279
+ "shape": [
280
+ 4096
281
+ ],
282
+ "dtype": "torch.bfloat16",
283
+ "sha256": "139e854517d7dcc3aec00f14bd727bc896037cae47eecedbf18d010a44e7f9b6"
284
+ },
285
+ {
286
+ "source_key": "model.visual.blocks.10.mlp.linear_fc1.weight",
287
+ "stored_key": "blocks.10.mlp.linear_fc1.weight",
288
+ "shape": [
289
+ 4096,
290
+ 1024
291
+ ],
292
+ "dtype": "torch.bfloat16",
293
+ "sha256": "8c572d053d961cc1b3c39b75ee9ee2bc8d32c2e2ef87f2d66ba4ecdcd9a024d3"
294
+ },
295
+ {
296
+ "source_key": "model.visual.blocks.10.mlp.linear_fc2.bias",
297
+ "stored_key": "blocks.10.mlp.linear_fc2.bias",
298
+ "shape": [
299
+ 1024
300
+ ],
301
+ "dtype": "torch.bfloat16",
302
+ "sha256": "7d50fdbd95fd91057efd6302c68549565b3cf98dfa3eba850dcafa9f45d5a3c5"
303
+ },
304
+ {
305
+ "source_key": "model.visual.blocks.10.mlp.linear_fc2.weight",
306
+ "stored_key": "blocks.10.mlp.linear_fc2.weight",
307
+ "shape": [
308
+ 1024,
309
+ 4096
310
+ ],
311
+ "dtype": "torch.bfloat16",
312
+ "sha256": "83ca6ba8955133aae1a2e54a6f10b422fcae03da01454d3341657db77ce13ec5"
313
+ },
314
+ {
315
+ "source_key": "model.visual.blocks.10.norm1.bias",
316
+ "stored_key": "blocks.10.norm1.bias",
317
+ "shape": [
318
+ 1024
319
+ ],
320
+ "dtype": "torch.bfloat16",
321
+ "sha256": "dc06e2e746f93ecf3e64cebf269143627a0ce9d7c3db17612bf74e91c7e2e4c9"
322
+ },
323
+ {
324
+ "source_key": "model.visual.blocks.10.norm1.weight",
325
+ "stored_key": "blocks.10.norm1.weight",
326
+ "shape": [
327
+ 1024
328
+ ],
329
+ "dtype": "torch.bfloat16",
330
+ "sha256": "719ffb80ffc4cf2f8e16e1c1df492c1db24edc03a600a09857d12023df3dd16e"
331
+ },
332
+ {
333
+ "source_key": "model.visual.blocks.10.norm2.bias",
334
+ "stored_key": "blocks.10.norm2.bias",
335
+ "shape": [
336
+ 1024
337
+ ],
338
+ "dtype": "torch.bfloat16",
339
+ "sha256": "b653a3ca18bd4cadba1c22baae51470529490e462b9698b683d73fa6aeae5112"
340
+ },
341
+ {
342
+ "source_key": "model.visual.blocks.10.norm2.weight",
343
+ "stored_key": "blocks.10.norm2.weight",
344
+ "shape": [
345
+ 1024
346
+ ],
347
+ "dtype": "torch.bfloat16",
348
+ "sha256": "9e212b794d6d3691b69856b810e1530ea40260d75f64849a6337ce7dd18745ca"
349
+ },
350
+ {
351
+ "source_key": "model.visual.blocks.11.attn.proj.bias",
352
+ "stored_key": "blocks.11.attn.proj.bias",
353
+ "shape": [
354
+ 1024
355
+ ],
356
+ "dtype": "torch.bfloat16",
357
+ "sha256": "535edf09375cf57feeda11cd58ae4fe5d3737895c989ca4228f4c524bded5410"
358
+ },
359
+ {
360
+ "source_key": "model.visual.blocks.11.attn.proj.weight",
361
+ "stored_key": "blocks.11.attn.proj.weight",
362
+ "shape": [
363
+ 1024,
364
+ 1024
365
+ ],
366
+ "dtype": "torch.bfloat16",
367
+ "sha256": "d36a71957345f59dc753c14ef8328ce053057bdc3c5c5dee12f974f686f566ed"
368
+ },
369
+ {
370
+ "source_key": "model.visual.blocks.11.attn.qkv.bias",
371
+ "stored_key": "blocks.11.attn.qkv.bias",
372
+ "shape": [
373
+ 3072
374
+ ],
375
+ "dtype": "torch.bfloat16",
376
+ "sha256": "4e1e5bf8a6f07c53127831d4d69c012fa69a3c93832943d2c95ec1d5bebbb118"
377
+ },
378
+ {
379
+ "source_key": "model.visual.blocks.11.attn.qkv.weight",
380
+ "stored_key": "blocks.11.attn.qkv.weight",
381
+ "shape": [
382
+ 3072,
383
+ 1024
384
+ ],
385
+ "dtype": "torch.bfloat16",
386
+ "sha256": "889e8de2ee3a15e6719594a2c60bafc2147bd01fee3c33192846bb6bb1e67243"
387
+ },
388
+ {
389
+ "source_key": "model.visual.blocks.11.mlp.linear_fc1.bias",
390
+ "stored_key": "blocks.11.mlp.linear_fc1.bias",
391
+ "shape": [
392
+ 4096
393
+ ],
394
+ "dtype": "torch.bfloat16",
395
+ "sha256": "e8aca63174bd64230ef4b8c378cb7f9d9f4a7bc5ee99092b8455df6f925c12fb"
396
+ },
397
+ {
398
+ "source_key": "model.visual.blocks.11.mlp.linear_fc1.weight",
399
+ "stored_key": "blocks.11.mlp.linear_fc1.weight",
400
+ "shape": [
401
+ 4096,
402
+ 1024
403
+ ],
404
+ "dtype": "torch.bfloat16",
405
+ "sha256": "945c321d140d345fff1ab00f658435ce41d451c7b34d83424e15ede0b81cadb9"
406
+ },
407
+ {
408
+ "source_key": "model.visual.blocks.11.mlp.linear_fc2.bias",
409
+ "stored_key": "blocks.11.mlp.linear_fc2.bias",
410
+ "shape": [
411
+ 1024
412
+ ],
413
+ "dtype": "torch.bfloat16",
414
+ "sha256": "5d45fd33d1c64f67ede91784d3efd0fa257990a730a2158e4edb193a102469fc"
415
+ },
416
+ {
417
+ "source_key": "model.visual.blocks.11.mlp.linear_fc2.weight",
418
+ "stored_key": "blocks.11.mlp.linear_fc2.weight",
419
+ "shape": [
420
+ 1024,
421
+ 4096
422
+ ],
423
+ "dtype": "torch.bfloat16",
424
+ "sha256": "2930219fe9770dc81932cb4535dbeb638dee12144eeef3db71cdaedf1fa65085"
425
+ },
426
+ {
427
+ "source_key": "model.visual.blocks.11.norm1.bias",
428
+ "stored_key": "blocks.11.norm1.bias",
429
+ "shape": [
430
+ 1024
431
+ ],
432
+ "dtype": "torch.bfloat16",
433
+ "sha256": "d4581447c32da3116b30c41c449810cc4cc84a7970bca02e615f90b2d4115217"
434
+ },
435
+ {
436
+ "source_key": "model.visual.blocks.11.norm1.weight",
437
+ "stored_key": "blocks.11.norm1.weight",
438
+ "shape": [
439
+ 1024
440
+ ],
441
+ "dtype": "torch.bfloat16",
442
+ "sha256": "1d2e5d2dce7cbeab8fc314d314c6647380c96f8ace9935f5cfff4b4345acc987"
443
+ },
444
+ {
445
+ "source_key": "model.visual.blocks.11.norm2.bias",
446
+ "stored_key": "blocks.11.norm2.bias",
447
+ "shape": [
448
+ 1024
449
+ ],
450
+ "dtype": "torch.bfloat16",
451
+ "sha256": "3418649580c5163edebfd05f860e39d0a35cf6038d469b10b55a360e612b5ae1"
452
+ },
453
+ {
454
+ "source_key": "model.visual.blocks.11.norm2.weight",
455
+ "stored_key": "blocks.11.norm2.weight",
456
+ "shape": [
457
+ 1024
458
+ ],
459
+ "dtype": "torch.bfloat16",
460
+ "sha256": "689cb527a1a646850a339bb03e4772f8c22da758610ff4f9ec5d8ef26e20fd8f"
461
+ },
462
+ {
463
+ "source_key": "model.visual.blocks.12.attn.proj.bias",
464
+ "stored_key": "blocks.12.attn.proj.bias",
465
+ "shape": [
466
+ 1024
467
+ ],
468
+ "dtype": "torch.bfloat16",
469
+ "sha256": "c7b571852801e9560968307fb74b42716175ad7074317a11c69a653e3cfcb90c"
470
+ },
471
+ {
472
+ "source_key": "model.visual.blocks.12.attn.proj.weight",
473
+ "stored_key": "blocks.12.attn.proj.weight",
474
+ "shape": [
475
+ 1024,
476
+ 1024
477
+ ],
478
+ "dtype": "torch.bfloat16",
479
+ "sha256": "66794d4cb4658fe29d02cfe07c712ee078f09d5b38a2269a8c173994ae9a2e4b"
480
+ },
481
+ {
482
+ "source_key": "model.visual.blocks.12.attn.qkv.bias",
483
+ "stored_key": "blocks.12.attn.qkv.bias",
484
+ "shape": [
485
+ 3072
486
+ ],
487
+ "dtype": "torch.bfloat16",
488
+ "sha256": "59e614d618c65fdc30ebc76b1d7fd0070aae08e8ddf166718b78f6f98c46199f"
489
+ },
490
+ {
491
+ "source_key": "model.visual.blocks.12.attn.qkv.weight",
492
+ "stored_key": "blocks.12.attn.qkv.weight",
493
+ "shape": [
494
+ 3072,
495
+ 1024
496
+ ],
497
+ "dtype": "torch.bfloat16",
498
+ "sha256": "d28bf809c5c142f96f00352170bd4211147570c6d7b50a3cb1c7a17a38f142df"
499
+ },
500
+ {
501
+ "source_key": "model.visual.blocks.12.mlp.linear_fc1.bias",
502
+ "stored_key": "blocks.12.mlp.linear_fc1.bias",
503
+ "shape": [
504
+ 4096
505
+ ],
506
+ "dtype": "torch.bfloat16",
507
+ "sha256": "03adf4517fd824e58ef9f3bf031b3731a02935648acbe834794da280308f79d0"
508
+ },
509
+ {
510
+ "source_key": "model.visual.blocks.12.mlp.linear_fc1.weight",
511
+ "stored_key": "blocks.12.mlp.linear_fc1.weight",
512
+ "shape": [
513
+ 4096,
514
+ 1024
515
+ ],
516
+ "dtype": "torch.bfloat16",
517
+ "sha256": "66971a356754963afe4f1ff1d946641cf4d6a31b1a40acf9578f435623686ad7"
518
+ },
519
+ {
520
+ "source_key": "model.visual.blocks.12.mlp.linear_fc2.bias",
521
+ "stored_key": "blocks.12.mlp.linear_fc2.bias",
522
+ "shape": [
523
+ 1024
524
+ ],
525
+ "dtype": "torch.bfloat16",
526
+ "sha256": "f58521a738b075b6dd324eba35b58060962f28e1bf790c54f3c1c23b4de7890e"
527
+ },
528
+ {
529
+ "source_key": "model.visual.blocks.12.mlp.linear_fc2.weight",
530
+ "stored_key": "blocks.12.mlp.linear_fc2.weight",
531
+ "shape": [
532
+ 1024,
533
+ 4096
534
+ ],
535
+ "dtype": "torch.bfloat16",
536
+ "sha256": "fcd763f6668a92b1678250b69e36e2c79a93ea333c6ccc79bccb8048db8f7f7f"
537
+ },
538
+ {
539
+ "source_key": "model.visual.blocks.12.norm1.bias",
540
+ "stored_key": "blocks.12.norm1.bias",
541
+ "shape": [
542
+ 1024
543
+ ],
544
+ "dtype": "torch.bfloat16",
545
+ "sha256": "55b488bec64e4932ed12e5e750846c7c74e1e0cb315a61b35994bcc840753e40"
546
+ },
547
+ {
548
+ "source_key": "model.visual.blocks.12.norm1.weight",
549
+ "stored_key": "blocks.12.norm1.weight",
550
+ "shape": [
551
+ 1024
552
+ ],
553
+ "dtype": "torch.bfloat16",
554
+ "sha256": "89d00ffecd4869aa34c0c652125a7ecf28f807f1d72ab54dde402596b3b70ba5"
555
+ },
556
+ {
557
+ "source_key": "model.visual.blocks.12.norm2.bias",
558
+ "stored_key": "blocks.12.norm2.bias",
559
+ "shape": [
560
+ 1024
561
+ ],
562
+ "dtype": "torch.bfloat16",
563
+ "sha256": "21a52d9ed6c14cc7b5be5623936bc8ce9c1772f1fa59b86b6c16ab5d30378d4d"
564
+ },
565
+ {
566
+ "source_key": "model.visual.blocks.12.norm2.weight",
567
+ "stored_key": "blocks.12.norm2.weight",
568
+ "shape": [
569
+ 1024
570
+ ],
571
+ "dtype": "torch.bfloat16",
572
+ "sha256": "e15ca91e098bc70d79fa65f39252d0184753746f907b6ad5dffc68cadbbcae19"
573
+ },
574
+ {
575
+ "source_key": "model.visual.blocks.13.attn.proj.bias",
576
+ "stored_key": "blocks.13.attn.proj.bias",
577
+ "shape": [
578
+ 1024
579
+ ],
580
+ "dtype": "torch.bfloat16",
581
+ "sha256": "cc45962b5cb888d7fe1cb1d61a478cfa127bbeedcce239ae5684061c56ffaa96"
582
+ },
583
+ {
584
+ "source_key": "model.visual.blocks.13.attn.proj.weight",
585
+ "stored_key": "blocks.13.attn.proj.weight",
586
+ "shape": [
587
+ 1024,
588
+ 1024
589
+ ],
590
+ "dtype": "torch.bfloat16",
591
+ "sha256": "d469689ed99301bf9aea5981c116a2ac5da38dba5d14ab5adb1fdf9f85e4ce4e"
592
+ },
593
+ {
594
+ "source_key": "model.visual.blocks.13.attn.qkv.bias",
595
+ "stored_key": "blocks.13.attn.qkv.bias",
596
+ "shape": [
597
+ 3072
598
+ ],
599
+ "dtype": "torch.bfloat16",
600
+ "sha256": "03bf87b59e1100efd55bfef2caa438efd931a25661644837fe2bf2b977ed2656"
601
+ },
602
+ {
603
+ "source_key": "model.visual.blocks.13.attn.qkv.weight",
604
+ "stored_key": "blocks.13.attn.qkv.weight",
605
+ "shape": [
606
+ 3072,
607
+ 1024
608
+ ],
609
+ "dtype": "torch.bfloat16",
610
+ "sha256": "9df7a0e82856323db50e5d4650225858b84fff6c17b06428cc18beee3251a492"
611
+ },
612
+ {
613
+ "source_key": "model.visual.blocks.13.mlp.linear_fc1.bias",
614
+ "stored_key": "blocks.13.mlp.linear_fc1.bias",
615
+ "shape": [
616
+ 4096
617
+ ],
618
+ "dtype": "torch.bfloat16",
619
+ "sha256": "75bdffb1019e93d21f224c813e7f6090565807a41f0d419720e27f6c88093861"
620
+ },
621
+ {
622
+ "source_key": "model.visual.blocks.13.mlp.linear_fc1.weight",
623
+ "stored_key": "blocks.13.mlp.linear_fc1.weight",
624
+ "shape": [
625
+ 4096,
626
+ 1024
627
+ ],
628
+ "dtype": "torch.bfloat16",
629
+ "sha256": "46ffd7228f1061198aea0cc1f2907e8a8f01975ffd84afb949d5808b442ce7e5"
630
+ },
631
+ {
632
+ "source_key": "model.visual.blocks.13.mlp.linear_fc2.bias",
633
+ "stored_key": "blocks.13.mlp.linear_fc2.bias",
634
+ "shape": [
635
+ 1024
636
+ ],
637
+ "dtype": "torch.bfloat16",
638
+ "sha256": "e837415b90547e07aa7c78db82d5c0af184cbd0eb21bb9241520e7c57f8cafd3"
639
+ },
640
+ {
641
+ "source_key": "model.visual.blocks.13.mlp.linear_fc2.weight",
642
+ "stored_key": "blocks.13.mlp.linear_fc2.weight",
643
+ "shape": [
644
+ 1024,
645
+ 4096
646
+ ],
647
+ "dtype": "torch.bfloat16",
648
+ "sha256": "d505906c14baf1c8e0e3e33cd8a7495afb8df9b1d187ceddc4d11327545a4cda"
649
+ },
650
+ {
651
+ "source_key": "model.visual.blocks.13.norm1.bias",
652
+ "stored_key": "blocks.13.norm1.bias",
653
+ "shape": [
654
+ 1024
655
+ ],
656
+ "dtype": "torch.bfloat16",
657
+ "sha256": "8f13ea991cb9be703f530c556e439753d2ced2596d8532fcb85811a22502ca72"
658
+ },
659
+ {
660
+ "source_key": "model.visual.blocks.13.norm1.weight",
661
+ "stored_key": "blocks.13.norm1.weight",
662
+ "shape": [
663
+ 1024
664
+ ],
665
+ "dtype": "torch.bfloat16",
666
+ "sha256": "14fcaa9177e529715de6d0fd8840bc3643126619555eb217eceadf989e515639"
667
+ },
668
+ {
669
+ "source_key": "model.visual.blocks.13.norm2.bias",
670
+ "stored_key": "blocks.13.norm2.bias",
671
+ "shape": [
672
+ 1024
673
+ ],
674
+ "dtype": "torch.bfloat16",
675
+ "sha256": "708302ab8178654bcf1bfcea3f2d3df37ec073b1a531dd15757e3a0807354f8a"
676
+ },
677
+ {
678
+ "source_key": "model.visual.blocks.13.norm2.weight",
679
+ "stored_key": "blocks.13.norm2.weight",
680
+ "shape": [
681
+ 1024
682
+ ],
683
+ "dtype": "torch.bfloat16",
684
+ "sha256": "9d7eb6708d8060db0f0614d46d04c7000cd1839c2e807806795b68c37153b45d"
685
+ },
686
+ {
687
+ "source_key": "model.visual.blocks.14.attn.proj.bias",
688
+ "stored_key": "blocks.14.attn.proj.bias",
689
+ "shape": [
690
+ 1024
691
+ ],
692
+ "dtype": "torch.bfloat16",
693
+ "sha256": "78b522a35237c3bdb0a1f5c203a2774138f837d092cc523071dd4688bab5c0e1"
694
+ },
695
+ {
696
+ "source_key": "model.visual.blocks.14.attn.proj.weight",
697
+ "stored_key": "blocks.14.attn.proj.weight",
698
+ "shape": [
699
+ 1024,
700
+ 1024
701
+ ],
702
+ "dtype": "torch.bfloat16",
703
+ "sha256": "495db4516a2878d1778c1539a3279e93d0122ff0e84444346070484b82b946f5"
704
+ },
705
+ {
706
+ "source_key": "model.visual.blocks.14.attn.qkv.bias",
707
+ "stored_key": "blocks.14.attn.qkv.bias",
708
+ "shape": [
709
+ 3072
710
+ ],
711
+ "dtype": "torch.bfloat16",
712
+ "sha256": "770fde88a2e7cdad9cafc19e437414a127410e29174d494819dd8d588997dfdc"
713
+ },
714
+ {
715
+ "source_key": "model.visual.blocks.14.attn.qkv.weight",
716
+ "stored_key": "blocks.14.attn.qkv.weight",
717
+ "shape": [
718
+ 3072,
719
+ 1024
720
+ ],
721
+ "dtype": "torch.bfloat16",
722
+ "sha256": "a654e0f77975ef02d00b5e2c13ccac4f4fc51be0dc947c9149123225a228f85b"
723
+ },
724
+ {
725
+ "source_key": "model.visual.blocks.14.mlp.linear_fc1.bias",
726
+ "stored_key": "blocks.14.mlp.linear_fc1.bias",
727
+ "shape": [
728
+ 4096
729
+ ],
730
+ "dtype": "torch.bfloat16",
731
+ "sha256": "96b5cedcc12924a57ebb92595fd14ef701ad1301d68485de48e2ed16df7cff76"
732
+ },
733
+ {
734
+ "source_key": "model.visual.blocks.14.mlp.linear_fc1.weight",
735
+ "stored_key": "blocks.14.mlp.linear_fc1.weight",
736
+ "shape": [
737
+ 4096,
738
+ 1024
739
+ ],
740
+ "dtype": "torch.bfloat16",
741
+ "sha256": "c14522925ec3cd97ede4b23b6700e07d136f3f5294b06aa2da94a50bed2dbf6f"
742
+ },
743
+ {
744
+ "source_key": "model.visual.blocks.14.mlp.linear_fc2.bias",
745
+ "stored_key": "blocks.14.mlp.linear_fc2.bias",
746
+ "shape": [
747
+ 1024
748
+ ],
749
+ "dtype": "torch.bfloat16",
750
+ "sha256": "234c7760d5a8eb130e2aaf1992e86904fd6bc76dfbfda85556429d4fedb3db95"
751
+ },
752
+ {
753
+ "source_key": "model.visual.blocks.14.mlp.linear_fc2.weight",
754
+ "stored_key": "blocks.14.mlp.linear_fc2.weight",
755
+ "shape": [
756
+ 1024,
757
+ 4096
758
+ ],
759
+ "dtype": "torch.bfloat16",
760
+ "sha256": "80983d2f8222003773fc0257bd81b781d653409b67ed956fe9a7c78db20c265e"
761
+ },
762
+ {
763
+ "source_key": "model.visual.blocks.14.norm1.bias",
764
+ "stored_key": "blocks.14.norm1.bias",
765
+ "shape": [
766
+ 1024
767
+ ],
768
+ "dtype": "torch.bfloat16",
769
+ "sha256": "ae1634ad682282777b579d0c6f082c013a572155ff91cf943548cfc8ab57dbd4"
770
+ },
771
+ {
772
+ "source_key": "model.visual.blocks.14.norm1.weight",
773
+ "stored_key": "blocks.14.norm1.weight",
774
+ "shape": [
775
+ 1024
776
+ ],
777
+ "dtype": "torch.bfloat16",
778
+ "sha256": "90ea83a611ffde5262720f085b6ef68c729a31d1874e93e8a266545a52c6574b"
779
+ },
780
+ {
781
+ "source_key": "model.visual.blocks.14.norm2.bias",
782
+ "stored_key": "blocks.14.norm2.bias",
783
+ "shape": [
784
+ 1024
785
+ ],
786
+ "dtype": "torch.bfloat16",
787
+ "sha256": "bb8469b7c041783c4c3aacd0a5322e6083eddc8b760914b7955c0a1590d83f67"
788
+ },
789
+ {
790
+ "source_key": "model.visual.blocks.14.norm2.weight",
791
+ "stored_key": "blocks.14.norm2.weight",
792
+ "shape": [
793
+ 1024
794
+ ],
795
+ "dtype": "torch.bfloat16",
796
+ "sha256": "63577cbee7225af4d2ba9a597d1b09a2f33f98f7a91c55d73c51de69dc40b228"
797
+ },
798
+ {
799
+ "source_key": "model.visual.blocks.15.attn.proj.bias",
800
+ "stored_key": "blocks.15.attn.proj.bias",
801
+ "shape": [
802
+ 1024
803
+ ],
804
+ "dtype": "torch.bfloat16",
805
+ "sha256": "d2fe021fbbeec2eae2c2e45cd019870330b50989595c58300a34a2fcdd73d87c"
806
+ },
807
+ {
808
+ "source_key": "model.visual.blocks.15.attn.proj.weight",
809
+ "stored_key": "blocks.15.attn.proj.weight",
810
+ "shape": [
811
+ 1024,
812
+ 1024
813
+ ],
814
+ "dtype": "torch.bfloat16",
815
+ "sha256": "4e475e6517c9d7054524601379b74d2c244f6a8bb8a9c9d82e5919a48e83adc1"
816
+ },
817
+ {
818
+ "source_key": "model.visual.blocks.15.attn.qkv.bias",
819
+ "stored_key": "blocks.15.attn.qkv.bias",
820
+ "shape": [
821
+ 3072
822
+ ],
823
+ "dtype": "torch.bfloat16",
824
+ "sha256": "c199d529f8f5609a1d321170146d751fd5f265f9f06403d21cdd0646f2a5735b"
825
+ },
826
+ {
827
+ "source_key": "model.visual.blocks.15.attn.qkv.weight",
828
+ "stored_key": "blocks.15.attn.qkv.weight",
829
+ "shape": [
830
+ 3072,
831
+ 1024
832
+ ],
833
+ "dtype": "torch.bfloat16",
834
+ "sha256": "2f0906f59c8d625695f4479fc600a6e93635599db07e7a4c36e74ffd7329c912"
835
+ },
836
+ {
837
+ "source_key": "model.visual.blocks.15.mlp.linear_fc1.bias",
838
+ "stored_key": "blocks.15.mlp.linear_fc1.bias",
839
+ "shape": [
840
+ 4096
841
+ ],
842
+ "dtype": "torch.bfloat16",
843
+ "sha256": "31c6ef245fc212fcbf57cfaae362f4ebabf4c5b1774e4b37351f45cc9c04309f"
844
+ },
845
+ {
846
+ "source_key": "model.visual.blocks.15.mlp.linear_fc1.weight",
847
+ "stored_key": "blocks.15.mlp.linear_fc1.weight",
848
+ "shape": [
849
+ 4096,
850
+ 1024
851
+ ],
852
+ "dtype": "torch.bfloat16",
853
+ "sha256": "c7d6c4fde5be6180fa11a3c0830c664bda4eb1b9d30494808f90dfce1243684a"
854
+ },
855
+ {
856
+ "source_key": "model.visual.blocks.15.mlp.linear_fc2.bias",
857
+ "stored_key": "blocks.15.mlp.linear_fc2.bias",
858
+ "shape": [
859
+ 1024
860
+ ],
861
+ "dtype": "torch.bfloat16",
862
+ "sha256": "e4873d62c1369b812cbbe58452ae2545384b9255d8e329b59072b806b583c610"
863
+ },
864
+ {
865
+ "source_key": "model.visual.blocks.15.mlp.linear_fc2.weight",
866
+ "stored_key": "blocks.15.mlp.linear_fc2.weight",
867
+ "shape": [
868
+ 1024,
869
+ 4096
870
+ ],
871
+ "dtype": "torch.bfloat16",
872
+ "sha256": "2ffec5982ec440803ebfad48d45c7154b292b9fa44bf0c9a74943d47b0910352"
873
+ },
874
+ {
875
+ "source_key": "model.visual.blocks.15.norm1.bias",
876
+ "stored_key": "blocks.15.norm1.bias",
877
+ "shape": [
878
+ 1024
879
+ ],
880
+ "dtype": "torch.bfloat16",
881
+ "sha256": "26afcefb186185a391fbd4f20ac7167806fe2b41fb5dd2cfac924d85dcaa29a2"
882
+ },
883
+ {
884
+ "source_key": "model.visual.blocks.15.norm1.weight",
885
+ "stored_key": "blocks.15.norm1.weight",
886
+ "shape": [
887
+ 1024
888
+ ],
889
+ "dtype": "torch.bfloat16",
890
+ "sha256": "9662b2e7b85e7fdd812feeda218a216999b49e302a4f56a519cee2702d53ebe5"
891
+ },
892
+ {
893
+ "source_key": "model.visual.blocks.15.norm2.bias",
894
+ "stored_key": "blocks.15.norm2.bias",
895
+ "shape": [
896
+ 1024
897
+ ],
898
+ "dtype": "torch.bfloat16",
899
+ "sha256": "c1d98bba835d65324080f1bf236913083d509b0e1eeb7c454660c9290788327a"
900
+ },
901
+ {
902
+ "source_key": "model.visual.blocks.15.norm2.weight",
903
+ "stored_key": "blocks.15.norm2.weight",
904
+ "shape": [
905
+ 1024
906
+ ],
907
+ "dtype": "torch.bfloat16",
908
+ "sha256": "42fd0802541e2a9b342e82ce403b093b916ed7f157f2a0e74fd2e8a2c64caa2e"
909
+ },
910
+ {
911
+ "source_key": "model.visual.blocks.16.attn.proj.bias",
912
+ "stored_key": "blocks.16.attn.proj.bias",
913
+ "shape": [
914
+ 1024
915
+ ],
916
+ "dtype": "torch.bfloat16",
917
+ "sha256": "67e303d1d3775f4c90619f2291cbfce8dd2db63cdac3f561d98b68170513da8e"
918
+ },
919
+ {
920
+ "source_key": "model.visual.blocks.16.attn.proj.weight",
921
+ "stored_key": "blocks.16.attn.proj.weight",
922
+ "shape": [
923
+ 1024,
924
+ 1024
925
+ ],
926
+ "dtype": "torch.bfloat16",
927
+ "sha256": "e035562d9b97d71cf1d3abf455ee350d3b9c716ca199f24b564c176ad30361d1"
928
+ },
929
+ {
930
+ "source_key": "model.visual.blocks.16.attn.qkv.bias",
931
+ "stored_key": "blocks.16.attn.qkv.bias",
932
+ "shape": [
933
+ 3072
934
+ ],
935
+ "dtype": "torch.bfloat16",
936
+ "sha256": "36cd7d7ce88a5574cabc98c53015513820f0f679bfe1b107ae4d19f523913269"
937
+ },
938
+ {
939
+ "source_key": "model.visual.blocks.16.attn.qkv.weight",
940
+ "stored_key": "blocks.16.attn.qkv.weight",
941
+ "shape": [
942
+ 3072,
943
+ 1024
944
+ ],
945
+ "dtype": "torch.bfloat16",
946
+ "sha256": "bdfed80d1ec1e6c6c6e428142250cd822e43de24002a869ad3386a256c65eff8"
947
+ },
948
+ {
949
+ "source_key": "model.visual.blocks.16.mlp.linear_fc1.bias",
950
+ "stored_key": "blocks.16.mlp.linear_fc1.bias",
951
+ "shape": [
952
+ 4096
953
+ ],
954
+ "dtype": "torch.bfloat16",
955
+ "sha256": "2679f3c0ab9440bab727659fe71545102d3d00ccf8ccce022af496c4a100439d"
956
+ },
957
+ {
958
+ "source_key": "model.visual.blocks.16.mlp.linear_fc1.weight",
959
+ "stored_key": "blocks.16.mlp.linear_fc1.weight",
960
+ "shape": [
961
+ 4096,
962
+ 1024
963
+ ],
964
+ "dtype": "torch.bfloat16",
965
+ "sha256": "cf9ae34a3e694231e5f53c49bc82d5b5165d1cef9659f58c7904504bfe379af3"
966
+ },
967
+ {
968
+ "source_key": "model.visual.blocks.16.mlp.linear_fc2.bias",
969
+ "stored_key": "blocks.16.mlp.linear_fc2.bias",
970
+ "shape": [
971
+ 1024
972
+ ],
973
+ "dtype": "torch.bfloat16",
974
+ "sha256": "9438d005f317c4914318b9ab88255f5702b29a809b3cf55e01e548a99eb29ea7"
975
+ },
976
+ {
977
+ "source_key": "model.visual.blocks.16.mlp.linear_fc2.weight",
978
+ "stored_key": "blocks.16.mlp.linear_fc2.weight",
979
+ "shape": [
980
+ 1024,
981
+ 4096
982
+ ],
983
+ "dtype": "torch.bfloat16",
984
+ "sha256": "7ec0e6ab70a768ae3ce247feafe6d3e9784a0c798ed0a08739f921124c5df224"
985
+ },
986
+ {
987
+ "source_key": "model.visual.blocks.16.norm1.bias",
988
+ "stored_key": "blocks.16.norm1.bias",
989
+ "shape": [
990
+ 1024
991
+ ],
992
+ "dtype": "torch.bfloat16",
993
+ "sha256": "922b5a50f265446fc1a9ab52f54220ad7143cfc97404b0311aab7c39409d7b04"
994
+ },
995
+ {
996
+ "source_key": "model.visual.blocks.16.norm1.weight",
997
+ "stored_key": "blocks.16.norm1.weight",
998
+ "shape": [
999
+ 1024
1000
+ ],
1001
+ "dtype": "torch.bfloat16",
1002
+ "sha256": "f1c8a5a85fd74b44e8a6aede6b86cdfe6147f65f5cb0fd63dd76380d0cd8aedf"
1003
+ },
1004
+ {
1005
+ "source_key": "model.visual.blocks.16.norm2.bias",
1006
+ "stored_key": "blocks.16.norm2.bias",
1007
+ "shape": [
1008
+ 1024
1009
+ ],
1010
+ "dtype": "torch.bfloat16",
1011
+ "sha256": "e881a4df26fb9f6972013b9d2ad61772532eec2d90d1959d93c0f6696db0dff8"
1012
+ },
1013
+ {
1014
+ "source_key": "model.visual.blocks.16.norm2.weight",
1015
+ "stored_key": "blocks.16.norm2.weight",
1016
+ "shape": [
1017
+ 1024
1018
+ ],
1019
+ "dtype": "torch.bfloat16",
1020
+ "sha256": "7a0044429ab55246a1001b025a619354c94e0cd6df5f521f16e632170ac9cbee"
1021
+ },
1022
+ {
1023
+ "source_key": "model.visual.blocks.17.attn.proj.bias",
1024
+ "stored_key": "blocks.17.attn.proj.bias",
1025
+ "shape": [
1026
+ 1024
1027
+ ],
1028
+ "dtype": "torch.bfloat16",
1029
+ "sha256": "ecc37835e3f012ba28a738cfabee4aa548c4e8c833227a19eadf3881f543778a"
1030
+ },
1031
+ {
1032
+ "source_key": "model.visual.blocks.17.attn.proj.weight",
1033
+ "stored_key": "blocks.17.attn.proj.weight",
1034
+ "shape": [
1035
+ 1024,
1036
+ 1024
1037
+ ],
1038
+ "dtype": "torch.bfloat16",
1039
+ "sha256": "1c4fb2f1482de43b399d8108a607207f7a258711664b083effceb7608c08fcdd"
1040
+ },
1041
+ {
1042
+ "source_key": "model.visual.blocks.17.attn.qkv.bias",
1043
+ "stored_key": "blocks.17.attn.qkv.bias",
1044
+ "shape": [
1045
+ 3072
1046
+ ],
1047
+ "dtype": "torch.bfloat16",
1048
+ "sha256": "3eec02c4679836ad3458f8ab0e8e9a5f3c0abc446a406c82845303703c977875"
1049
+ },
1050
+ {
1051
+ "source_key": "model.visual.blocks.17.attn.qkv.weight",
1052
+ "stored_key": "blocks.17.attn.qkv.weight",
1053
+ "shape": [
1054
+ 3072,
1055
+ 1024
1056
+ ],
1057
+ "dtype": "torch.bfloat16",
1058
+ "sha256": "a76df2bc1117e32356129da359319caeccf348f48ef418198d209ab6251ac4e6"
1059
+ },
1060
+ {
1061
+ "source_key": "model.visual.blocks.17.mlp.linear_fc1.bias",
1062
+ "stored_key": "blocks.17.mlp.linear_fc1.bias",
1063
+ "shape": [
1064
+ 4096
1065
+ ],
1066
+ "dtype": "torch.bfloat16",
1067
+ "sha256": "f4e4905f09f3fba726cc3729e0afb674553f856d3e48a00e2627ae449905f002"
1068
+ },
1069
+ {
1070
+ "source_key": "model.visual.blocks.17.mlp.linear_fc1.weight",
1071
+ "stored_key": "blocks.17.mlp.linear_fc1.weight",
1072
+ "shape": [
1073
+ 4096,
1074
+ 1024
1075
+ ],
1076
+ "dtype": "torch.bfloat16",
1077
+ "sha256": "a653be31b0da49a30dd573359ce5bd3d6ae1001258db8916935e8778d44988b7"
1078
+ },
1079
+ {
1080
+ "source_key": "model.visual.blocks.17.mlp.linear_fc2.bias",
1081
+ "stored_key": "blocks.17.mlp.linear_fc2.bias",
1082
+ "shape": [
1083
+ 1024
1084
+ ],
1085
+ "dtype": "torch.bfloat16",
1086
+ "sha256": "53f3fb23e9187cb3acf7fa2fce202178bfbba27e441a797a0e273715cd5d02fe"
1087
+ },
1088
+ {
1089
+ "source_key": "model.visual.blocks.17.mlp.linear_fc2.weight",
1090
+ "stored_key": "blocks.17.mlp.linear_fc2.weight",
1091
+ "shape": [
1092
+ 1024,
1093
+ 4096
1094
+ ],
1095
+ "dtype": "torch.bfloat16",
1096
+ "sha256": "35a0bef5571381a0ddddbace0bff928e6e85b692050a9fa41ee875184d95cd67"
1097
+ },
1098
+ {
1099
+ "source_key": "model.visual.blocks.17.norm1.bias",
1100
+ "stored_key": "blocks.17.norm1.bias",
1101
+ "shape": [
1102
+ 1024
1103
+ ],
1104
+ "dtype": "torch.bfloat16",
1105
+ "sha256": "a70dbee2047483ef903515abd41def95857c933997574836d69b96434d6b2002"
1106
+ },
1107
+ {
1108
+ "source_key": "model.visual.blocks.17.norm1.weight",
1109
+ "stored_key": "blocks.17.norm1.weight",
1110
+ "shape": [
1111
+ 1024
1112
+ ],
1113
+ "dtype": "torch.bfloat16",
1114
+ "sha256": "6a249e8a02afd9e50b6a8a82df8980bea1b7f930bf8e081213d56939b9f82347"
1115
+ },
1116
+ {
1117
+ "source_key": "model.visual.blocks.17.norm2.bias",
1118
+ "stored_key": "blocks.17.norm2.bias",
1119
+ "shape": [
1120
+ 1024
1121
+ ],
1122
+ "dtype": "torch.bfloat16",
1123
+ "sha256": "67c6caeb0198d4c00066e3d08a81d30757f8c85521ade900e424291e5b34c94c"
1124
+ },
1125
+ {
1126
+ "source_key": "model.visual.blocks.17.norm2.weight",
1127
+ "stored_key": "blocks.17.norm2.weight",
1128
+ "shape": [
1129
+ 1024
1130
+ ],
1131
+ "dtype": "torch.bfloat16",
1132
+ "sha256": "d07112b08b9563293a057bf47f60f22343569ca52ad56d04d101cad8d81aad7b"
1133
+ },
1134
+ {
1135
+ "source_key": "model.visual.blocks.18.attn.proj.bias",
1136
+ "stored_key": "blocks.18.attn.proj.bias",
1137
+ "shape": [
1138
+ 1024
1139
+ ],
1140
+ "dtype": "torch.bfloat16",
1141
+ "sha256": "d189414425632c6b37d3c3f8bdff10a16519caca1b9cee8ea5d3d504bc6e8953"
1142
+ },
1143
+ {
1144
+ "source_key": "model.visual.blocks.18.attn.proj.weight",
1145
+ "stored_key": "blocks.18.attn.proj.weight",
1146
+ "shape": [
1147
+ 1024,
1148
+ 1024
1149
+ ],
1150
+ "dtype": "torch.bfloat16",
1151
+ "sha256": "70aa98a9e2db36a5e16d485f5a808bbf562bf7d55c2e6b6316cbaf543cc3ff07"
1152
+ },
1153
+ {
1154
+ "source_key": "model.visual.blocks.18.attn.qkv.bias",
1155
+ "stored_key": "blocks.18.attn.qkv.bias",
1156
+ "shape": [
1157
+ 3072
1158
+ ],
1159
+ "dtype": "torch.bfloat16",
1160
+ "sha256": "e0db722d012df5d9af6e8bbf6f79260dd0746e1fcc4667c021ef4a7a0c49a552"
1161
+ },
1162
+ {
1163
+ "source_key": "model.visual.blocks.18.attn.qkv.weight",
1164
+ "stored_key": "blocks.18.attn.qkv.weight",
1165
+ "shape": [
1166
+ 3072,
1167
+ 1024
1168
+ ],
1169
+ "dtype": "torch.bfloat16",
1170
+ "sha256": "518bf9d6418e421d00c5d3908f4c44186131afc074a813c43ddd6030a529d6c4"
1171
+ },
1172
+ {
1173
+ "source_key": "model.visual.blocks.18.mlp.linear_fc1.bias",
1174
+ "stored_key": "blocks.18.mlp.linear_fc1.bias",
1175
+ "shape": [
1176
+ 4096
1177
+ ],
1178
+ "dtype": "torch.bfloat16",
1179
+ "sha256": "9d56fe5066357e6869c6f7d95855bb6f519bd0f5d1ed26b113392a6b1b0a5e30"
1180
+ },
1181
+ {
1182
+ "source_key": "model.visual.blocks.18.mlp.linear_fc1.weight",
1183
+ "stored_key": "blocks.18.mlp.linear_fc1.weight",
1184
+ "shape": [
1185
+ 4096,
1186
+ 1024
1187
+ ],
1188
+ "dtype": "torch.bfloat16",
1189
+ "sha256": "23b3edd3baf70fa1db318d60b8d97ac163472ce5a29efd6a5121680c1368a652"
1190
+ },
1191
+ {
1192
+ "source_key": "model.visual.blocks.18.mlp.linear_fc2.bias",
1193
+ "stored_key": "blocks.18.mlp.linear_fc2.bias",
1194
+ "shape": [
1195
+ 1024
1196
+ ],
1197
+ "dtype": "torch.bfloat16",
1198
+ "sha256": "cf4371fe8d5dac5e6907d6b18d2d23d842cdefdfcf3a6af4dec684b39db1f7aa"
1199
+ },
1200
+ {
1201
+ "source_key": "model.visual.blocks.18.mlp.linear_fc2.weight",
1202
+ "stored_key": "blocks.18.mlp.linear_fc2.weight",
1203
+ "shape": [
1204
+ 1024,
1205
+ 4096
1206
+ ],
1207
+ "dtype": "torch.bfloat16",
1208
+ "sha256": "12f8fb5d6d417e24eff8df41b03d154e1f86ed68158f1d62ca81347a6f61a0f7"
1209
+ },
1210
+ {
1211
+ "source_key": "model.visual.blocks.18.norm1.bias",
1212
+ "stored_key": "blocks.18.norm1.bias",
1213
+ "shape": [
1214
+ 1024
1215
+ ],
1216
+ "dtype": "torch.bfloat16",
1217
+ "sha256": "2dd09a0b0b575c6fb2539920e3b1a3d30748d7808e24415e5d248b83c274ba79"
1218
+ },
1219
+ {
1220
+ "source_key": "model.visual.blocks.18.norm1.weight",
1221
+ "stored_key": "blocks.18.norm1.weight",
1222
+ "shape": [
1223
+ 1024
1224
+ ],
1225
+ "dtype": "torch.bfloat16",
1226
+ "sha256": "11440904da22b07846a01a67f36111d4df1cdafbbe83b5a263fe80510b2b8be0"
1227
+ },
1228
+ {
1229
+ "source_key": "model.visual.blocks.18.norm2.bias",
1230
+ "stored_key": "blocks.18.norm2.bias",
1231
+ "shape": [
1232
+ 1024
1233
+ ],
1234
+ "dtype": "torch.bfloat16",
1235
+ "sha256": "3a5ea949a98848992b5622056ea1255ed379bb0319583a6d7bebc645caa4bce4"
1236
+ },
1237
+ {
1238
+ "source_key": "model.visual.blocks.18.norm2.weight",
1239
+ "stored_key": "blocks.18.norm2.weight",
1240
+ "shape": [
1241
+ 1024
1242
+ ],
1243
+ "dtype": "torch.bfloat16",
1244
+ "sha256": "a552debe1a2ddfb61489569868979b5c6241d92296e0bec833d59994de494f00"
1245
+ },
1246
+ {
1247
+ "source_key": "model.visual.blocks.19.attn.proj.bias",
1248
+ "stored_key": "blocks.19.attn.proj.bias",
1249
+ "shape": [
1250
+ 1024
1251
+ ],
1252
+ "dtype": "torch.bfloat16",
1253
+ "sha256": "f2be32c53bdcb95dc24995aee285447268badc536ba0beb9f9533560e750f6f1"
1254
+ },
1255
+ {
1256
+ "source_key": "model.visual.blocks.19.attn.proj.weight",
1257
+ "stored_key": "blocks.19.attn.proj.weight",
1258
+ "shape": [
1259
+ 1024,
1260
+ 1024
1261
+ ],
1262
+ "dtype": "torch.bfloat16",
1263
+ "sha256": "a879c4ab75e252fce71be40052e2541ce36f6259808cbda1db243494500af3af"
1264
+ },
1265
+ {
1266
+ "source_key": "model.visual.blocks.19.attn.qkv.bias",
1267
+ "stored_key": "blocks.19.attn.qkv.bias",
1268
+ "shape": [
1269
+ 3072
1270
+ ],
1271
+ "dtype": "torch.bfloat16",
1272
+ "sha256": "63eab9f5da59cfc96453e09f1ddb8bb5961d6a5166a07f21965b371083e75d3a"
1273
+ },
1274
+ {
1275
+ "source_key": "model.visual.blocks.19.attn.qkv.weight",
1276
+ "stored_key": "blocks.19.attn.qkv.weight",
1277
+ "shape": [
1278
+ 3072,
1279
+ 1024
1280
+ ],
1281
+ "dtype": "torch.bfloat16",
1282
+ "sha256": "cf8284358d735cd78b61075796b001de9b5ae65dc0d6c353000e5bcbabf61566"
1283
+ },
1284
+ {
1285
+ "source_key": "model.visual.blocks.19.mlp.linear_fc1.bias",
1286
+ "stored_key": "blocks.19.mlp.linear_fc1.bias",
1287
+ "shape": [
1288
+ 4096
1289
+ ],
1290
+ "dtype": "torch.bfloat16",
1291
+ "sha256": "c4231bf87f87eab9d97b8a843f9df38daedf450f5e438c21809f941fb3e63754"
1292
+ },
1293
+ {
1294
+ "source_key": "model.visual.blocks.19.mlp.linear_fc1.weight",
1295
+ "stored_key": "blocks.19.mlp.linear_fc1.weight",
1296
+ "shape": [
1297
+ 4096,
1298
+ 1024
1299
+ ],
1300
+ "dtype": "torch.bfloat16",
1301
+ "sha256": "37758c33670d2393fe70dad407d77ebb7f6ee36d9f18a152dfdc58eec2bf1d9d"
1302
+ },
1303
+ {
1304
+ "source_key": "model.visual.blocks.19.mlp.linear_fc2.bias",
1305
+ "stored_key": "blocks.19.mlp.linear_fc2.bias",
1306
+ "shape": [
1307
+ 1024
1308
+ ],
1309
+ "dtype": "torch.bfloat16",
1310
+ "sha256": "9d6d57bd7e8efa67b604dca0d8138a096e297569f752b839dec81cd45ca1528a"
1311
+ },
1312
+ {
1313
+ "source_key": "model.visual.blocks.19.mlp.linear_fc2.weight",
1314
+ "stored_key": "blocks.19.mlp.linear_fc2.weight",
1315
+ "shape": [
1316
+ 1024,
1317
+ 4096
1318
+ ],
1319
+ "dtype": "torch.bfloat16",
1320
+ "sha256": "efa8961ef4c18b8c1af6471ea4fb0fed1649735f8798b371cd5ffb9f31844755"
1321
+ },
1322
+ {
1323
+ "source_key": "model.visual.blocks.19.norm1.bias",
1324
+ "stored_key": "blocks.19.norm1.bias",
1325
+ "shape": [
1326
+ 1024
1327
+ ],
1328
+ "dtype": "torch.bfloat16",
1329
+ "sha256": "14bf205dcbeff508fba49b85da9e3b110a228582ba925c27a32b2dc1f9852934"
1330
+ },
1331
+ {
1332
+ "source_key": "model.visual.blocks.19.norm1.weight",
1333
+ "stored_key": "blocks.19.norm1.weight",
1334
+ "shape": [
1335
+ 1024
1336
+ ],
1337
+ "dtype": "torch.bfloat16",
1338
+ "sha256": "bb1acdb28398581fd4d2872985498bad6f70658f24c8b82ddcf2f292dd4ee6e5"
1339
+ },
1340
+ {
1341
+ "source_key": "model.visual.blocks.19.norm2.bias",
1342
+ "stored_key": "blocks.19.norm2.bias",
1343
+ "shape": [
1344
+ 1024
1345
+ ],
1346
+ "dtype": "torch.bfloat16",
1347
+ "sha256": "6f4d0d1aa976eca8422fef218f36e233585f7a71d007921c7e0b7a7fa7cb20b7"
1348
+ },
1349
+ {
1350
+ "source_key": "model.visual.blocks.19.norm2.weight",
1351
+ "stored_key": "blocks.19.norm2.weight",
1352
+ "shape": [
1353
+ 1024
1354
+ ],
1355
+ "dtype": "torch.bfloat16",
1356
+ "sha256": "6c6b4fff1eec770266f86d6302fc538aaa3e99a055a64e5ac25a890a9d105cf1"
1357
+ },
1358
+ {
1359
+ "source_key": "model.visual.blocks.2.attn.proj.bias",
1360
+ "stored_key": "blocks.2.attn.proj.bias",
1361
+ "shape": [
1362
+ 1024
1363
+ ],
1364
+ "dtype": "torch.bfloat16",
1365
+ "sha256": "d15d8952876c443d2831242bbca7e2085b66fdb7cc9f028f86c4ce0c54ed10eb"
1366
+ },
1367
+ {
1368
+ "source_key": "model.visual.blocks.2.attn.proj.weight",
1369
+ "stored_key": "blocks.2.attn.proj.weight",
1370
+ "shape": [
1371
+ 1024,
1372
+ 1024
1373
+ ],
1374
+ "dtype": "torch.bfloat16",
1375
+ "sha256": "095bfe7575b6611583f259f734b85c29d3943ba3b03fb664f9101bd4454f8b29"
1376
+ },
1377
+ {
1378
+ "source_key": "model.visual.blocks.2.attn.qkv.bias",
1379
+ "stored_key": "blocks.2.attn.qkv.bias",
1380
+ "shape": [
1381
+ 3072
1382
+ ],
1383
+ "dtype": "torch.bfloat16",
1384
+ "sha256": "36da5adcd26a03140fce35ef1fe043fd05d1e8f08be447e70f2695f637f581d5"
1385
+ },
1386
+ {
1387
+ "source_key": "model.visual.blocks.2.attn.qkv.weight",
1388
+ "stored_key": "blocks.2.attn.qkv.weight",
1389
+ "shape": [
1390
+ 3072,
1391
+ 1024
1392
+ ],
1393
+ "dtype": "torch.bfloat16",
1394
+ "sha256": "c03817aa9ba2c70252eecc07da07e86a533392b91a507190c2eb2eb3603005bc"
1395
+ },
1396
+ {
1397
+ "source_key": "model.visual.blocks.2.mlp.linear_fc1.bias",
1398
+ "stored_key": "blocks.2.mlp.linear_fc1.bias",
1399
+ "shape": [
1400
+ 4096
1401
+ ],
1402
+ "dtype": "torch.bfloat16",
1403
+ "sha256": "565baff198a031a862174d097fddf2762ffb5d3e380546e96b9a4b19233b0e0d"
1404
+ },
1405
+ {
1406
+ "source_key": "model.visual.blocks.2.mlp.linear_fc1.weight",
1407
+ "stored_key": "blocks.2.mlp.linear_fc1.weight",
1408
+ "shape": [
1409
+ 4096,
1410
+ 1024
1411
+ ],
1412
+ "dtype": "torch.bfloat16",
1413
+ "sha256": "d537438f5898a64f7ac3175cd934cdc946e53deaf37c4b36ad085102c4bcf705"
1414
+ },
1415
+ {
1416
+ "source_key": "model.visual.blocks.2.mlp.linear_fc2.bias",
1417
+ "stored_key": "blocks.2.mlp.linear_fc2.bias",
1418
+ "shape": [
1419
+ 1024
1420
+ ],
1421
+ "dtype": "torch.bfloat16",
1422
+ "sha256": "f0c2fab04bffd3bc60617339b7d0e4750b395b7af93aba14efa04c4f4376ebb8"
1423
+ },
1424
+ {
1425
+ "source_key": "model.visual.blocks.2.mlp.linear_fc2.weight",
1426
+ "stored_key": "blocks.2.mlp.linear_fc2.weight",
1427
+ "shape": [
1428
+ 1024,
1429
+ 4096
1430
+ ],
1431
+ "dtype": "torch.bfloat16",
1432
+ "sha256": "4e1ce0b6b4d4e730b1fb1729e9ddd3f81fb1061b010aa08992be59bbf01991b7"
1433
+ },
1434
+ {
1435
+ "source_key": "model.visual.blocks.2.norm1.bias",
1436
+ "stored_key": "blocks.2.norm1.bias",
1437
+ "shape": [
1438
+ 1024
1439
+ ],
1440
+ "dtype": "torch.bfloat16",
1441
+ "sha256": "d53bb37eb27be8e8367972ee20877e0ff53c9c26d13a9796f756b35a80314f65"
1442
+ },
1443
+ {
1444
+ "source_key": "model.visual.blocks.2.norm1.weight",
1445
+ "stored_key": "blocks.2.norm1.weight",
1446
+ "shape": [
1447
+ 1024
1448
+ ],
1449
+ "dtype": "torch.bfloat16",
1450
+ "sha256": "53eb7e4d525bc90817a878bec7e9ab2d17ba965fad21194cc0e8825542de4dd1"
1451
+ },
1452
+ {
1453
+ "source_key": "model.visual.blocks.2.norm2.bias",
1454
+ "stored_key": "blocks.2.norm2.bias",
1455
+ "shape": [
1456
+ 1024
1457
+ ],
1458
+ "dtype": "torch.bfloat16",
1459
+ "sha256": "de42b54380f8d4bc114c8d329c2fb86caf33d3920665fbaab7f8b041e856e499"
1460
+ },
1461
+ {
1462
+ "source_key": "model.visual.blocks.2.norm2.weight",
1463
+ "stored_key": "blocks.2.norm2.weight",
1464
+ "shape": [
1465
+ 1024
1466
+ ],
1467
+ "dtype": "torch.bfloat16",
1468
+ "sha256": "72301bded64b8f8586d447c2a747a7237e4be7caed8e23a902278c3d8b23b0c9"
1469
+ },
1470
+ {
1471
+ "source_key": "model.visual.blocks.20.attn.proj.bias",
1472
+ "stored_key": "blocks.20.attn.proj.bias",
1473
+ "shape": [
1474
+ 1024
1475
+ ],
1476
+ "dtype": "torch.bfloat16",
1477
+ "sha256": "5e364ea41ac34ef692a35f2deb56f452cdbe752fd60affe6c2d212efa3c7156a"
1478
+ },
1479
+ {
1480
+ "source_key": "model.visual.blocks.20.attn.proj.weight",
1481
+ "stored_key": "blocks.20.attn.proj.weight",
1482
+ "shape": [
1483
+ 1024,
1484
+ 1024
1485
+ ],
1486
+ "dtype": "torch.bfloat16",
1487
+ "sha256": "8261f057e2b0acbd86da2618f384e4849164272b96de798760403240a883f0bf"
1488
+ },
1489
+ {
1490
+ "source_key": "model.visual.blocks.20.attn.qkv.bias",
1491
+ "stored_key": "blocks.20.attn.qkv.bias",
1492
+ "shape": [
1493
+ 3072
1494
+ ],
1495
+ "dtype": "torch.bfloat16",
1496
+ "sha256": "c85dfcc71c85ca886ffbd410b375bdce1ec02eadd0320c5cb90e9af91babefcb"
1497
+ },
1498
+ {
1499
+ "source_key": "model.visual.blocks.20.attn.qkv.weight",
1500
+ "stored_key": "blocks.20.attn.qkv.weight",
1501
+ "shape": [
1502
+ 3072,
1503
+ 1024
1504
+ ],
1505
+ "dtype": "torch.bfloat16",
1506
+ "sha256": "78934f651d247405b38476dc01be1ea2692ed605d0f18af1a726c9defed93fde"
1507
+ },
1508
+ {
1509
+ "source_key": "model.visual.blocks.20.mlp.linear_fc1.bias",
1510
+ "stored_key": "blocks.20.mlp.linear_fc1.bias",
1511
+ "shape": [
1512
+ 4096
1513
+ ],
1514
+ "dtype": "torch.bfloat16",
1515
+ "sha256": "cd52afabacae8dd7cba4133d47a3c1efc317cbb3baacf3c2b70736ea1b8930ed"
1516
+ },
1517
+ {
1518
+ "source_key": "model.visual.blocks.20.mlp.linear_fc1.weight",
1519
+ "stored_key": "blocks.20.mlp.linear_fc1.weight",
1520
+ "shape": [
1521
+ 4096,
1522
+ 1024
1523
+ ],
1524
+ "dtype": "torch.bfloat16",
1525
+ "sha256": "2793660b0fb055c039adcb4e885613a625336e17b8fea48c21607c0cb35ebdc6"
1526
+ },
1527
+ {
1528
+ "source_key": "model.visual.blocks.20.mlp.linear_fc2.bias",
1529
+ "stored_key": "blocks.20.mlp.linear_fc2.bias",
1530
+ "shape": [
1531
+ 1024
1532
+ ],
1533
+ "dtype": "torch.bfloat16",
1534
+ "sha256": "90f48412166b57b1f83e92b9da46c72898fd8b00daed46cf5124609aa0dcc518"
1535
+ },
1536
+ {
1537
+ "source_key": "model.visual.blocks.20.mlp.linear_fc2.weight",
1538
+ "stored_key": "blocks.20.mlp.linear_fc2.weight",
1539
+ "shape": [
1540
+ 1024,
1541
+ 4096
1542
+ ],
1543
+ "dtype": "torch.bfloat16",
1544
+ "sha256": "7bb8e17178071bd63796cdb123b125b2306c304da1540eed4be466b3702e37c0"
1545
+ },
1546
+ {
1547
+ "source_key": "model.visual.blocks.20.norm1.bias",
1548
+ "stored_key": "blocks.20.norm1.bias",
1549
+ "shape": [
1550
+ 1024
1551
+ ],
1552
+ "dtype": "torch.bfloat16",
1553
+ "sha256": "6f32885d3347215cf127e6de8ad993f2d075537af31b22038e2e4b16b8457aff"
1554
+ },
1555
+ {
1556
+ "source_key": "model.visual.blocks.20.norm1.weight",
1557
+ "stored_key": "blocks.20.norm1.weight",
1558
+ "shape": [
1559
+ 1024
1560
+ ],
1561
+ "dtype": "torch.bfloat16",
1562
+ "sha256": "dc8bb31922829c4336dc609c6bae2cbf895e4d926f264d5169fcb1078b2e73e1"
1563
+ },
1564
+ {
1565
+ "source_key": "model.visual.blocks.20.norm2.bias",
1566
+ "stored_key": "blocks.20.norm2.bias",
1567
+ "shape": [
1568
+ 1024
1569
+ ],
1570
+ "dtype": "torch.bfloat16",
1571
+ "sha256": "e7313b6bcd24a7a6a6709898f96d68cd77f5ed28f16f4d2facdb4b47355d66b6"
1572
+ },
1573
+ {
1574
+ "source_key": "model.visual.blocks.20.norm2.weight",
1575
+ "stored_key": "blocks.20.norm2.weight",
1576
+ "shape": [
1577
+ 1024
1578
+ ],
1579
+ "dtype": "torch.bfloat16",
1580
+ "sha256": "cd576cf8d2cb289107a5f36d200dcd6181e9b2aff110eb67fa8501e714c061fe"
1581
+ },
1582
+ {
1583
+ "source_key": "model.visual.blocks.21.attn.proj.bias",
1584
+ "stored_key": "blocks.21.attn.proj.bias",
1585
+ "shape": [
1586
+ 1024
1587
+ ],
1588
+ "dtype": "torch.bfloat16",
1589
+ "sha256": "1caea704b06a98ccf5b07e66800475e28a29a6a30545bb232d5e02bbceee5074"
1590
+ },
1591
+ {
1592
+ "source_key": "model.visual.blocks.21.attn.proj.weight",
1593
+ "stored_key": "blocks.21.attn.proj.weight",
1594
+ "shape": [
1595
+ 1024,
1596
+ 1024
1597
+ ],
1598
+ "dtype": "torch.bfloat16",
1599
+ "sha256": "e42ccabdb9a2a400a22742d1fbd81e4413827972188944e975f9eda2f75f2d3f"
1600
+ },
1601
+ {
1602
+ "source_key": "model.visual.blocks.21.attn.qkv.bias",
1603
+ "stored_key": "blocks.21.attn.qkv.bias",
1604
+ "shape": [
1605
+ 3072
1606
+ ],
1607
+ "dtype": "torch.bfloat16",
1608
+ "sha256": "289f99f76968b3c9256921fe61947b9ad1a57e04b98362847036dcd1970b6221"
1609
+ },
1610
+ {
1611
+ "source_key": "model.visual.blocks.21.attn.qkv.weight",
1612
+ "stored_key": "blocks.21.attn.qkv.weight",
1613
+ "shape": [
1614
+ 3072,
1615
+ 1024
1616
+ ],
1617
+ "dtype": "torch.bfloat16",
1618
+ "sha256": "4f1e6dfb1828b2cba6e232a4eb65d87205e929952ddc4e76b7596de082156202"
1619
+ },
1620
+ {
1621
+ "source_key": "model.visual.blocks.21.mlp.linear_fc1.bias",
1622
+ "stored_key": "blocks.21.mlp.linear_fc1.bias",
1623
+ "shape": [
1624
+ 4096
1625
+ ],
1626
+ "dtype": "torch.bfloat16",
1627
+ "sha256": "acd9bbcd7f65285b6e72632f67714857f8c7b793114dd50d02a5222308be0e60"
1628
+ },
1629
+ {
1630
+ "source_key": "model.visual.blocks.21.mlp.linear_fc1.weight",
1631
+ "stored_key": "blocks.21.mlp.linear_fc1.weight",
1632
+ "shape": [
1633
+ 4096,
1634
+ 1024
1635
+ ],
1636
+ "dtype": "torch.bfloat16",
1637
+ "sha256": "aa45c52155c439153ac1c1d44d122ff6f8a9eeb7b410fdc7514755fd70cd2e3b"
1638
+ },
1639
+ {
1640
+ "source_key": "model.visual.blocks.21.mlp.linear_fc2.bias",
1641
+ "stored_key": "blocks.21.mlp.linear_fc2.bias",
1642
+ "shape": [
1643
+ 1024
1644
+ ],
1645
+ "dtype": "torch.bfloat16",
1646
+ "sha256": "bdf54d1be829f1a6e72a2969505a4ee41972377259c2de877adfb71a090a5f55"
1647
+ },
1648
+ {
1649
+ "source_key": "model.visual.blocks.21.mlp.linear_fc2.weight",
1650
+ "stored_key": "blocks.21.mlp.linear_fc2.weight",
1651
+ "shape": [
1652
+ 1024,
1653
+ 4096
1654
+ ],
1655
+ "dtype": "torch.bfloat16",
1656
+ "sha256": "e091320ca94b4b06c33af4bff2f71df5a9c414ff542d5737c38629fc997df5c5"
1657
+ },
1658
+ {
1659
+ "source_key": "model.visual.blocks.21.norm1.bias",
1660
+ "stored_key": "blocks.21.norm1.bias",
1661
+ "shape": [
1662
+ 1024
1663
+ ],
1664
+ "dtype": "torch.bfloat16",
1665
+ "sha256": "66ff6f25051c4b7ade96f574c50ff7cc394524a3a77471d3472c7dbf579b5632"
1666
+ },
1667
+ {
1668
+ "source_key": "model.visual.blocks.21.norm1.weight",
1669
+ "stored_key": "blocks.21.norm1.weight",
1670
+ "shape": [
1671
+ 1024
1672
+ ],
1673
+ "dtype": "torch.bfloat16",
1674
+ "sha256": "721cea6b2e9d821ea3687de6a41f9d51da0105aaa37adc9f0707a2144fbbee21"
1675
+ },
1676
+ {
1677
+ "source_key": "model.visual.blocks.21.norm2.bias",
1678
+ "stored_key": "blocks.21.norm2.bias",
1679
+ "shape": [
1680
+ 1024
1681
+ ],
1682
+ "dtype": "torch.bfloat16",
1683
+ "sha256": "074ae062e8572f4468f98d3e87d3e6e8108726b3ddf47f8046e501791aad58ce"
1684
+ },
1685
+ {
1686
+ "source_key": "model.visual.blocks.21.norm2.weight",
1687
+ "stored_key": "blocks.21.norm2.weight",
1688
+ "shape": [
1689
+ 1024
1690
+ ],
1691
+ "dtype": "torch.bfloat16",
1692
+ "sha256": "85efd3a842a203524b24bcfa2c4e8dbe6a66c4c15e701db048cb5172700c10e4"
1693
+ },
1694
+ {
1695
+ "source_key": "model.visual.blocks.22.attn.proj.bias",
1696
+ "stored_key": "blocks.22.attn.proj.bias",
1697
+ "shape": [
1698
+ 1024
1699
+ ],
1700
+ "dtype": "torch.bfloat16",
1701
+ "sha256": "5aef38229be345796c155c7fe6571ceadf55ff05e797a888a1c55edd04a6255b"
1702
+ },
1703
+ {
1704
+ "source_key": "model.visual.blocks.22.attn.proj.weight",
1705
+ "stored_key": "blocks.22.attn.proj.weight",
1706
+ "shape": [
1707
+ 1024,
1708
+ 1024
1709
+ ],
1710
+ "dtype": "torch.bfloat16",
1711
+ "sha256": "dec478193ac6059b2b38fb0c3c6d9967d690164c7baf082dae1076d7f9cd1eb5"
1712
+ },
1713
+ {
1714
+ "source_key": "model.visual.blocks.22.attn.qkv.bias",
1715
+ "stored_key": "blocks.22.attn.qkv.bias",
1716
+ "shape": [
1717
+ 3072
1718
+ ],
1719
+ "dtype": "torch.bfloat16",
1720
+ "sha256": "90a8fe246319b8b330ca3fad67b2ff43da0eed2421b5872ac8ade749114da75d"
1721
+ },
1722
+ {
1723
+ "source_key": "model.visual.blocks.22.attn.qkv.weight",
1724
+ "stored_key": "blocks.22.attn.qkv.weight",
1725
+ "shape": [
1726
+ 3072,
1727
+ 1024
1728
+ ],
1729
+ "dtype": "torch.bfloat16",
1730
+ "sha256": "56a5c9eaa7061ee04b50ef49593c432e1cf3266e631c64bd239fec464128fd20"
1731
+ },
1732
+ {
1733
+ "source_key": "model.visual.blocks.22.mlp.linear_fc1.bias",
1734
+ "stored_key": "blocks.22.mlp.linear_fc1.bias",
1735
+ "shape": [
1736
+ 4096
1737
+ ],
1738
+ "dtype": "torch.bfloat16",
1739
+ "sha256": "f6bb1804338a46e001f04de162efa193d2fe06078fedd0cf89b7097e3642dacc"
1740
+ },
1741
+ {
1742
+ "source_key": "model.visual.blocks.22.mlp.linear_fc1.weight",
1743
+ "stored_key": "blocks.22.mlp.linear_fc1.weight",
1744
+ "shape": [
1745
+ 4096,
1746
+ 1024
1747
+ ],
1748
+ "dtype": "torch.bfloat16",
1749
+ "sha256": "8878d6c70d1223efb67eb560531575084607b373f9cf3093dec62cd80cfea01b"
1750
+ },
1751
+ {
1752
+ "source_key": "model.visual.blocks.22.mlp.linear_fc2.bias",
1753
+ "stored_key": "blocks.22.mlp.linear_fc2.bias",
1754
+ "shape": [
1755
+ 1024
1756
+ ],
1757
+ "dtype": "torch.bfloat16",
1758
+ "sha256": "cb0343ab54df7a316159c2ea6822e1a2219f78541d3f56006a53c2eb8d8a8760"
1759
+ },
1760
+ {
1761
+ "source_key": "model.visual.blocks.22.mlp.linear_fc2.weight",
1762
+ "stored_key": "blocks.22.mlp.linear_fc2.weight",
1763
+ "shape": [
1764
+ 1024,
1765
+ 4096
1766
+ ],
1767
+ "dtype": "torch.bfloat16",
1768
+ "sha256": "4d23ba22952fbc540a74d41269121283dd63a26d6299233275845e52af58b179"
1769
+ },
1770
+ {
1771
+ "source_key": "model.visual.blocks.22.norm1.bias",
1772
+ "stored_key": "blocks.22.norm1.bias",
1773
+ "shape": [
1774
+ 1024
1775
+ ],
1776
+ "dtype": "torch.bfloat16",
1777
+ "sha256": "7b658a00cf46afe84ca7e42b32282a40a31954eedf54b425c62b4b372c23aa1d"
1778
+ },
1779
+ {
1780
+ "source_key": "model.visual.blocks.22.norm1.weight",
1781
+ "stored_key": "blocks.22.norm1.weight",
1782
+ "shape": [
1783
+ 1024
1784
+ ],
1785
+ "dtype": "torch.bfloat16",
1786
+ "sha256": "613a10552c93f56ff5bac07a7bf3cb471f0946c63637f6e8fd8b08087d8d799b"
1787
+ },
1788
+ {
1789
+ "source_key": "model.visual.blocks.22.norm2.bias",
1790
+ "stored_key": "blocks.22.norm2.bias",
1791
+ "shape": [
1792
+ 1024
1793
+ ],
1794
+ "dtype": "torch.bfloat16",
1795
+ "sha256": "70fb9cdc8f888e9949cb3945439182cd90dbc92e06320a1faacac82791f57ded"
1796
+ },
1797
+ {
1798
+ "source_key": "model.visual.blocks.22.norm2.weight",
1799
+ "stored_key": "blocks.22.norm2.weight",
1800
+ "shape": [
1801
+ 1024
1802
+ ],
1803
+ "dtype": "torch.bfloat16",
1804
+ "sha256": "5d1d84f6f59a9db6fbb3bd3c1ed3e897d5de96ac11ddb41fc6d3e34a9056786e"
1805
+ },
1806
+ {
1807
+ "source_key": "model.visual.blocks.23.attn.proj.bias",
1808
+ "stored_key": "blocks.23.attn.proj.bias",
1809
+ "shape": [
1810
+ 1024
1811
+ ],
1812
+ "dtype": "torch.bfloat16",
1813
+ "sha256": "355fbbe43395049f2206ced956fe07361dfad2e8a6a80cf5c741c58081a0b2d1"
1814
+ },
1815
+ {
1816
+ "source_key": "model.visual.blocks.23.attn.proj.weight",
1817
+ "stored_key": "blocks.23.attn.proj.weight",
1818
+ "shape": [
1819
+ 1024,
1820
+ 1024
1821
+ ],
1822
+ "dtype": "torch.bfloat16",
1823
+ "sha256": "90770d5378a963bc3b69f61f04d617e6fc0086034fae3e285086353ee8c134e4"
1824
+ },
1825
+ {
1826
+ "source_key": "model.visual.blocks.23.attn.qkv.bias",
1827
+ "stored_key": "blocks.23.attn.qkv.bias",
1828
+ "shape": [
1829
+ 3072
1830
+ ],
1831
+ "dtype": "torch.bfloat16",
1832
+ "sha256": "c76e9e6223e2d4b98a8921eaf31a9a8f6473dbd7bb1a29bef26336999a3dddbd"
1833
+ },
1834
+ {
1835
+ "source_key": "model.visual.blocks.23.attn.qkv.weight",
1836
+ "stored_key": "blocks.23.attn.qkv.weight",
1837
+ "shape": [
1838
+ 3072,
1839
+ 1024
1840
+ ],
1841
+ "dtype": "torch.bfloat16",
1842
+ "sha256": "babd730e9ca3ee7a8d8a4e03aece2c0f003034fbc834983d7f93cfc9da49b56e"
1843
+ },
1844
+ {
1845
+ "source_key": "model.visual.blocks.23.mlp.linear_fc1.bias",
1846
+ "stored_key": "blocks.23.mlp.linear_fc1.bias",
1847
+ "shape": [
1848
+ 4096
1849
+ ],
1850
+ "dtype": "torch.bfloat16",
1851
+ "sha256": "9d5a25961f99eca0f561435b25614206fba993bd814f2366503980b79bc87fbd"
1852
+ },
1853
+ {
1854
+ "source_key": "model.visual.blocks.23.mlp.linear_fc1.weight",
1855
+ "stored_key": "blocks.23.mlp.linear_fc1.weight",
1856
+ "shape": [
1857
+ 4096,
1858
+ 1024
1859
+ ],
1860
+ "dtype": "torch.bfloat16",
1861
+ "sha256": "3f37bebcf28909d9e5739dd9c018824e0415989b30dfac31615cee2082ca9c64"
1862
+ },
1863
+ {
1864
+ "source_key": "model.visual.blocks.23.mlp.linear_fc2.bias",
1865
+ "stored_key": "blocks.23.mlp.linear_fc2.bias",
1866
+ "shape": [
1867
+ 1024
1868
+ ],
1869
+ "dtype": "torch.bfloat16",
1870
+ "sha256": "797dd716b3b55db78566e29159ee8da7d2ca74a84f4e784cbf5d4b120db5f668"
1871
+ },
1872
+ {
1873
+ "source_key": "model.visual.blocks.23.mlp.linear_fc2.weight",
1874
+ "stored_key": "blocks.23.mlp.linear_fc2.weight",
1875
+ "shape": [
1876
+ 1024,
1877
+ 4096
1878
+ ],
1879
+ "dtype": "torch.bfloat16",
1880
+ "sha256": "c099129a931c240c719768b43d59566b0c02b94e07cb90c9dd676701180e8729"
1881
+ },
1882
+ {
1883
+ "source_key": "model.visual.blocks.23.norm1.bias",
1884
+ "stored_key": "blocks.23.norm1.bias",
1885
+ "shape": [
1886
+ 1024
1887
+ ],
1888
+ "dtype": "torch.bfloat16",
1889
+ "sha256": "15dc7c26cd78d8e259bd592071c9923287fa643ed77988b18f6437c8de9fc774"
1890
+ },
1891
+ {
1892
+ "source_key": "model.visual.blocks.23.norm1.weight",
1893
+ "stored_key": "blocks.23.norm1.weight",
1894
+ "shape": [
1895
+ 1024
1896
+ ],
1897
+ "dtype": "torch.bfloat16",
1898
+ "sha256": "dab60e5b1828f0f302fb85e166b1a3b335d10b6a164f398dad8651dd3df9633c"
1899
+ },
1900
+ {
1901
+ "source_key": "model.visual.blocks.23.norm2.bias",
1902
+ "stored_key": "blocks.23.norm2.bias",
1903
+ "shape": [
1904
+ 1024
1905
+ ],
1906
+ "dtype": "torch.bfloat16",
1907
+ "sha256": "8ce06c37f923af0456619d405017ecc7f0468a828cd33de69faa5476c1800b35"
1908
+ },
1909
+ {
1910
+ "source_key": "model.visual.blocks.23.norm2.weight",
1911
+ "stored_key": "blocks.23.norm2.weight",
1912
+ "shape": [
1913
+ 1024
1914
+ ],
1915
+ "dtype": "torch.bfloat16",
1916
+ "sha256": "0b47bcbb28867fdf2cf58da7255b5a766c3676b9a3a73483a738b8c08b072580"
1917
+ },
1918
+ {
1919
+ "source_key": "model.visual.blocks.3.attn.proj.bias",
1920
+ "stored_key": "blocks.3.attn.proj.bias",
1921
+ "shape": [
1922
+ 1024
1923
+ ],
1924
+ "dtype": "torch.bfloat16",
1925
+ "sha256": "2906df962e6f718f2889e0733692498d828d7a47c921daac0e41f830947266a2"
1926
+ },
1927
+ {
1928
+ "source_key": "model.visual.blocks.3.attn.proj.weight",
1929
+ "stored_key": "blocks.3.attn.proj.weight",
1930
+ "shape": [
1931
+ 1024,
1932
+ 1024
1933
+ ],
1934
+ "dtype": "torch.bfloat16",
1935
+ "sha256": "536b81ffcc6ab4f3f5c873c098385af8da3bb5a8d49c27d204136aa19b2b3802"
1936
+ },
1937
+ {
1938
+ "source_key": "model.visual.blocks.3.attn.qkv.bias",
1939
+ "stored_key": "blocks.3.attn.qkv.bias",
1940
+ "shape": [
1941
+ 3072
1942
+ ],
1943
+ "dtype": "torch.bfloat16",
1944
+ "sha256": "b092c8f5454519288f5bf47b5e10a44a9e2a6079b51e0e10e1a7ecd2654e5150"
1945
+ },
1946
+ {
1947
+ "source_key": "model.visual.blocks.3.attn.qkv.weight",
1948
+ "stored_key": "blocks.3.attn.qkv.weight",
1949
+ "shape": [
1950
+ 3072,
1951
+ 1024
1952
+ ],
1953
+ "dtype": "torch.bfloat16",
1954
+ "sha256": "7e76a81b955c26dfbcdd872abbd120078896123ba925da5e37adb67b9b1a77e0"
1955
+ },
1956
+ {
1957
+ "source_key": "model.visual.blocks.3.mlp.linear_fc1.bias",
1958
+ "stored_key": "blocks.3.mlp.linear_fc1.bias",
1959
+ "shape": [
1960
+ 4096
1961
+ ],
1962
+ "dtype": "torch.bfloat16",
1963
+ "sha256": "00077141432334861eeb1edb44fabe79d881937364039f0a657462c8d0af5aad"
1964
+ },
1965
+ {
1966
+ "source_key": "model.visual.blocks.3.mlp.linear_fc1.weight",
1967
+ "stored_key": "blocks.3.mlp.linear_fc1.weight",
1968
+ "shape": [
1969
+ 4096,
1970
+ 1024
1971
+ ],
1972
+ "dtype": "torch.bfloat16",
1973
+ "sha256": "869faedbf272b0187c5d014f80c315117e5b8aaa2762da869b80a881095a640b"
1974
+ },
1975
+ {
1976
+ "source_key": "model.visual.blocks.3.mlp.linear_fc2.bias",
1977
+ "stored_key": "blocks.3.mlp.linear_fc2.bias",
1978
+ "shape": [
1979
+ 1024
1980
+ ],
1981
+ "dtype": "torch.bfloat16",
1982
+ "sha256": "76ff805b1edefae7b3bea63fe7efffb0a29fb2c5f92e815b42a01ea7395281d7"
1983
+ },
1984
+ {
1985
+ "source_key": "model.visual.blocks.3.mlp.linear_fc2.weight",
1986
+ "stored_key": "blocks.3.mlp.linear_fc2.weight",
1987
+ "shape": [
1988
+ 1024,
1989
+ 4096
1990
+ ],
1991
+ "dtype": "torch.bfloat16",
1992
+ "sha256": "317615bd5284e068e2bd750b9ca2ceea093508b308358185b47736c50a20869f"
1993
+ },
1994
+ {
1995
+ "source_key": "model.visual.blocks.3.norm1.bias",
1996
+ "stored_key": "blocks.3.norm1.bias",
1997
+ "shape": [
1998
+ 1024
1999
+ ],
2000
+ "dtype": "torch.bfloat16",
2001
+ "sha256": "d7b9392b156e16b2e61c9e1899e5ad227af1a0d9a59f06da3dd4bcd32f7af5ae"
2002
+ },
2003
+ {
2004
+ "source_key": "model.visual.blocks.3.norm1.weight",
2005
+ "stored_key": "blocks.3.norm1.weight",
2006
+ "shape": [
2007
+ 1024
2008
+ ],
2009
+ "dtype": "torch.bfloat16",
2010
+ "sha256": "30181cc25c28fe2b05ae78d48b511ac7e37da9e0687bc78c1fb413a33118a688"
2011
+ },
2012
+ {
2013
+ "source_key": "model.visual.blocks.3.norm2.bias",
2014
+ "stored_key": "blocks.3.norm2.bias",
2015
+ "shape": [
2016
+ 1024
2017
+ ],
2018
+ "dtype": "torch.bfloat16",
2019
+ "sha256": "2f27bb2f660f96da6f925290fa4356fdfaca3f8dc1c0f59f3baeb53ab1e98c4e"
2020
+ },
2021
+ {
2022
+ "source_key": "model.visual.blocks.3.norm2.weight",
2023
+ "stored_key": "blocks.3.norm2.weight",
2024
+ "shape": [
2025
+ 1024
2026
+ ],
2027
+ "dtype": "torch.bfloat16",
2028
+ "sha256": "347d99b48dc2daddf26f0b142cd0e147b1e7001715f1f8174fa4763f949d2a78"
2029
+ },
2030
+ {
2031
+ "source_key": "model.visual.blocks.4.attn.proj.bias",
2032
+ "stored_key": "blocks.4.attn.proj.bias",
2033
+ "shape": [
2034
+ 1024
2035
+ ],
2036
+ "dtype": "torch.bfloat16",
2037
+ "sha256": "4bea9d5edb4899dc3123838b610f43ffe3ab847781c09ee75ac96c5f5543a21e"
2038
+ },
2039
+ {
2040
+ "source_key": "model.visual.blocks.4.attn.proj.weight",
2041
+ "stored_key": "blocks.4.attn.proj.weight",
2042
+ "shape": [
2043
+ 1024,
2044
+ 1024
2045
+ ],
2046
+ "dtype": "torch.bfloat16",
2047
+ "sha256": "f949f8b84f16ab4c8d35549ba06865c6f9aac90e7ba4871e4fdfe9267c9e851b"
2048
+ },
2049
+ {
2050
+ "source_key": "model.visual.blocks.4.attn.qkv.bias",
2051
+ "stored_key": "blocks.4.attn.qkv.bias",
2052
+ "shape": [
2053
+ 3072
2054
+ ],
2055
+ "dtype": "torch.bfloat16",
2056
+ "sha256": "56255d2c8fbfbc0e8c730e4ffa92edcb520b7414a519d838e8fc6886c57efa4b"
2057
+ },
2058
+ {
2059
+ "source_key": "model.visual.blocks.4.attn.qkv.weight",
2060
+ "stored_key": "blocks.4.attn.qkv.weight",
2061
+ "shape": [
2062
+ 3072,
2063
+ 1024
2064
+ ],
2065
+ "dtype": "torch.bfloat16",
2066
+ "sha256": "78e94955c572d74473a1bf8dc18804171e3925ab785d8716ede4dbc9b1387d23"
2067
+ },
2068
+ {
2069
+ "source_key": "model.visual.blocks.4.mlp.linear_fc1.bias",
2070
+ "stored_key": "blocks.4.mlp.linear_fc1.bias",
2071
+ "shape": [
2072
+ 4096
2073
+ ],
2074
+ "dtype": "torch.bfloat16",
2075
+ "sha256": "d861c45d68ff93f6d5c5292c0f55914318d1e18c3c538f74f08fd31cc27126e0"
2076
+ },
2077
+ {
2078
+ "source_key": "model.visual.blocks.4.mlp.linear_fc1.weight",
2079
+ "stored_key": "blocks.4.mlp.linear_fc1.weight",
2080
+ "shape": [
2081
+ 4096,
2082
+ 1024
2083
+ ],
2084
+ "dtype": "torch.bfloat16",
2085
+ "sha256": "75a6adefb4c89cca46e1b1749b21e0a06d60b66c2107667df93758b96cc529b6"
2086
+ },
2087
+ {
2088
+ "source_key": "model.visual.blocks.4.mlp.linear_fc2.bias",
2089
+ "stored_key": "blocks.4.mlp.linear_fc2.bias",
2090
+ "shape": [
2091
+ 1024
2092
+ ],
2093
+ "dtype": "torch.bfloat16",
2094
+ "sha256": "a0f49ebdb81855883f202562f011256cdd27f11e45fe5d6ae6ed7ade365a70b3"
2095
+ },
2096
+ {
2097
+ "source_key": "model.visual.blocks.4.mlp.linear_fc2.weight",
2098
+ "stored_key": "blocks.4.mlp.linear_fc2.weight",
2099
+ "shape": [
2100
+ 1024,
2101
+ 4096
2102
+ ],
2103
+ "dtype": "torch.bfloat16",
2104
+ "sha256": "084f11d67c1473dd6d46965b3339a7eca94bc7e21ac7663f2b15e7ea573d56dd"
2105
+ },
2106
+ {
2107
+ "source_key": "model.visual.blocks.4.norm1.bias",
2108
+ "stored_key": "blocks.4.norm1.bias",
2109
+ "shape": [
2110
+ 1024
2111
+ ],
2112
+ "dtype": "torch.bfloat16",
2113
+ "sha256": "e3f21e806c215a36471ce272360df57bc533fa797372005f9f31a1046da52fa5"
2114
+ },
2115
+ {
2116
+ "source_key": "model.visual.blocks.4.norm1.weight",
2117
+ "stored_key": "blocks.4.norm1.weight",
2118
+ "shape": [
2119
+ 1024
2120
+ ],
2121
+ "dtype": "torch.bfloat16",
2122
+ "sha256": "912fa73a0040ccdd66e5cd3700c36cb879278d0285235a884739af4bdacf9873"
2123
+ },
2124
+ {
2125
+ "source_key": "model.visual.blocks.4.norm2.bias",
2126
+ "stored_key": "blocks.4.norm2.bias",
2127
+ "shape": [
2128
+ 1024
2129
+ ],
2130
+ "dtype": "torch.bfloat16",
2131
+ "sha256": "ecf54355a1afce4cb71c6a5e0dd1e89f3dc453dc6120ab7d1b43843f986af168"
2132
+ },
2133
+ {
2134
+ "source_key": "model.visual.blocks.4.norm2.weight",
2135
+ "stored_key": "blocks.4.norm2.weight",
2136
+ "shape": [
2137
+ 1024
2138
+ ],
2139
+ "dtype": "torch.bfloat16",
2140
+ "sha256": "53538fafbccf36456e0c38893a9ab5ed86ddc2eb467612525b02e05c025f3dc4"
2141
+ },
2142
+ {
2143
+ "source_key": "model.visual.blocks.5.attn.proj.bias",
2144
+ "stored_key": "blocks.5.attn.proj.bias",
2145
+ "shape": [
2146
+ 1024
2147
+ ],
2148
+ "dtype": "torch.bfloat16",
2149
+ "sha256": "50b4b834515e66d84b93a23baab3f88a82f26644a3f4e94251575ee88323aacb"
2150
+ },
2151
+ {
2152
+ "source_key": "model.visual.blocks.5.attn.proj.weight",
2153
+ "stored_key": "blocks.5.attn.proj.weight",
2154
+ "shape": [
2155
+ 1024,
2156
+ 1024
2157
+ ],
2158
+ "dtype": "torch.bfloat16",
2159
+ "sha256": "8b380198c4c4816879ee8207196d0618379f245d192cb3c7c7d3f3fd81470ac6"
2160
+ },
2161
+ {
2162
+ "source_key": "model.visual.blocks.5.attn.qkv.bias",
2163
+ "stored_key": "blocks.5.attn.qkv.bias",
2164
+ "shape": [
2165
+ 3072
2166
+ ],
2167
+ "dtype": "torch.bfloat16",
2168
+ "sha256": "1c3e40dbb84ff086f14d0a2b318962300d64c7122735f50119e1931aea92c13b"
2169
+ },
2170
+ {
2171
+ "source_key": "model.visual.blocks.5.attn.qkv.weight",
2172
+ "stored_key": "blocks.5.attn.qkv.weight",
2173
+ "shape": [
2174
+ 3072,
2175
+ 1024
2176
+ ],
2177
+ "dtype": "torch.bfloat16",
2178
+ "sha256": "f715a2bee024faff4d13049b53ca1d45ce8988054342756c7a69cfbc75726738"
2179
+ },
2180
+ {
2181
+ "source_key": "model.visual.blocks.5.mlp.linear_fc1.bias",
2182
+ "stored_key": "blocks.5.mlp.linear_fc1.bias",
2183
+ "shape": [
2184
+ 4096
2185
+ ],
2186
+ "dtype": "torch.bfloat16",
2187
+ "sha256": "84af37bb9a0c3bb0cc5c04d15aedb31343d2ab1c7b122c3e8063d82b88512937"
2188
+ },
2189
+ {
2190
+ "source_key": "model.visual.blocks.5.mlp.linear_fc1.weight",
2191
+ "stored_key": "blocks.5.mlp.linear_fc1.weight",
2192
+ "shape": [
2193
+ 4096,
2194
+ 1024
2195
+ ],
2196
+ "dtype": "torch.bfloat16",
2197
+ "sha256": "1b182ba753b9519751073b22f9c7e3eb85b24d7824e265ee73a2660df2c857e5"
2198
+ },
2199
+ {
2200
+ "source_key": "model.visual.blocks.5.mlp.linear_fc2.bias",
2201
+ "stored_key": "blocks.5.mlp.linear_fc2.bias",
2202
+ "shape": [
2203
+ 1024
2204
+ ],
2205
+ "dtype": "torch.bfloat16",
2206
+ "sha256": "a0a50dda36aa78bd562cb741cf9f55550de9b73f7567eceb00a7bc0a7585c5eb"
2207
+ },
2208
+ {
2209
+ "source_key": "model.visual.blocks.5.mlp.linear_fc2.weight",
2210
+ "stored_key": "blocks.5.mlp.linear_fc2.weight",
2211
+ "shape": [
2212
+ 1024,
2213
+ 4096
2214
+ ],
2215
+ "dtype": "torch.bfloat16",
2216
+ "sha256": "f0e0b9cc801daa75f4ac8969b0fae4b8ef54dadf1c7d0a20a829f9df805535f2"
2217
+ },
2218
+ {
2219
+ "source_key": "model.visual.blocks.5.norm1.bias",
2220
+ "stored_key": "blocks.5.norm1.bias",
2221
+ "shape": [
2222
+ 1024
2223
+ ],
2224
+ "dtype": "torch.bfloat16",
2225
+ "sha256": "6b667b92dea407df9c8a9fdbad542d35e7ac12cbb8d67a167e3534ff65ad32ad"
2226
+ },
2227
+ {
2228
+ "source_key": "model.visual.blocks.5.norm1.weight",
2229
+ "stored_key": "blocks.5.norm1.weight",
2230
+ "shape": [
2231
+ 1024
2232
+ ],
2233
+ "dtype": "torch.bfloat16",
2234
+ "sha256": "27a9d0f5055cc243eb1e3dfe3af5dd60391535dc36c1a7ffbfa72108dc59eb45"
2235
+ },
2236
+ {
2237
+ "source_key": "model.visual.blocks.5.norm2.bias",
2238
+ "stored_key": "blocks.5.norm2.bias",
2239
+ "shape": [
2240
+ 1024
2241
+ ],
2242
+ "dtype": "torch.bfloat16",
2243
+ "sha256": "61452e3ca053ba408bdbd70c510e8d99316bdc0f9e6daa269905e604e427eb27"
2244
+ },
2245
+ {
2246
+ "source_key": "model.visual.blocks.5.norm2.weight",
2247
+ "stored_key": "blocks.5.norm2.weight",
2248
+ "shape": [
2249
+ 1024
2250
+ ],
2251
+ "dtype": "torch.bfloat16",
2252
+ "sha256": "c2503f4b1f8d9759cd85e58d733e8908f54eb06a1333588ebcbf0d1da80856ac"
2253
+ },
2254
+ {
2255
+ "source_key": "model.visual.blocks.6.attn.proj.bias",
2256
+ "stored_key": "blocks.6.attn.proj.bias",
2257
+ "shape": [
2258
+ 1024
2259
+ ],
2260
+ "dtype": "torch.bfloat16",
2261
+ "sha256": "ee534892c5c917664794ea432c5cc5a59b468e87b5dfa9c14d457524918628ea"
2262
+ },
2263
+ {
2264
+ "source_key": "model.visual.blocks.6.attn.proj.weight",
2265
+ "stored_key": "blocks.6.attn.proj.weight",
2266
+ "shape": [
2267
+ 1024,
2268
+ 1024
2269
+ ],
2270
+ "dtype": "torch.bfloat16",
2271
+ "sha256": "da2dcfca947f1777341ce986767b895cff1bbb554722c9548fd73a0de3b13751"
2272
+ },
2273
+ {
2274
+ "source_key": "model.visual.blocks.6.attn.qkv.bias",
2275
+ "stored_key": "blocks.6.attn.qkv.bias",
2276
+ "shape": [
2277
+ 3072
2278
+ ],
2279
+ "dtype": "torch.bfloat16",
2280
+ "sha256": "d251bd07c049f8f044dbe0246ad431a5d1e722d54e769eaf077779283c5d6339"
2281
+ },
2282
+ {
2283
+ "source_key": "model.visual.blocks.6.attn.qkv.weight",
2284
+ "stored_key": "blocks.6.attn.qkv.weight",
2285
+ "shape": [
2286
+ 3072,
2287
+ 1024
2288
+ ],
2289
+ "dtype": "torch.bfloat16",
2290
+ "sha256": "bc7bc88703a0654465e2f24f01fcf0455d3c65aee5babfe02c2e2861c3396884"
2291
+ },
2292
+ {
2293
+ "source_key": "model.visual.blocks.6.mlp.linear_fc1.bias",
2294
+ "stored_key": "blocks.6.mlp.linear_fc1.bias",
2295
+ "shape": [
2296
+ 4096
2297
+ ],
2298
+ "dtype": "torch.bfloat16",
2299
+ "sha256": "c91fcf391e724944303017c76130899b7f071f126812c5c5e5c89038f1bcdba9"
2300
+ },
2301
+ {
2302
+ "source_key": "model.visual.blocks.6.mlp.linear_fc1.weight",
2303
+ "stored_key": "blocks.6.mlp.linear_fc1.weight",
2304
+ "shape": [
2305
+ 4096,
2306
+ 1024
2307
+ ],
2308
+ "dtype": "torch.bfloat16",
2309
+ "sha256": "aec0b03fb7e670015eb4d4652add7f4594322a71b5e75bcd81a44e539a3d0a45"
2310
+ },
2311
+ {
2312
+ "source_key": "model.visual.blocks.6.mlp.linear_fc2.bias",
2313
+ "stored_key": "blocks.6.mlp.linear_fc2.bias",
2314
+ "shape": [
2315
+ 1024
2316
+ ],
2317
+ "dtype": "torch.bfloat16",
2318
+ "sha256": "0273437e510930b85a59540e217609e597bf7ad4dff1cb89c162d9b9c9dd5516"
2319
+ },
2320
+ {
2321
+ "source_key": "model.visual.blocks.6.mlp.linear_fc2.weight",
2322
+ "stored_key": "blocks.6.mlp.linear_fc2.weight",
2323
+ "shape": [
2324
+ 1024,
2325
+ 4096
2326
+ ],
2327
+ "dtype": "torch.bfloat16",
2328
+ "sha256": "6ced53d3e857dcaf11230b09be55d5b7d130283fa652e5dfce6f009f469dab0e"
2329
+ },
2330
+ {
2331
+ "source_key": "model.visual.blocks.6.norm1.bias",
2332
+ "stored_key": "blocks.6.norm1.bias",
2333
+ "shape": [
2334
+ 1024
2335
+ ],
2336
+ "dtype": "torch.bfloat16",
2337
+ "sha256": "02b18baf1c0c0440276913acf7eba813ae28aaee3aef255ef30846d0877a0210"
2338
+ },
2339
+ {
2340
+ "source_key": "model.visual.blocks.6.norm1.weight",
2341
+ "stored_key": "blocks.6.norm1.weight",
2342
+ "shape": [
2343
+ 1024
2344
+ ],
2345
+ "dtype": "torch.bfloat16",
2346
+ "sha256": "6eb5a08f270365b7c434c500a7c1d77f5cfbfef4d09ea5626a120c97768517f4"
2347
+ },
2348
+ {
2349
+ "source_key": "model.visual.blocks.6.norm2.bias",
2350
+ "stored_key": "blocks.6.norm2.bias",
2351
+ "shape": [
2352
+ 1024
2353
+ ],
2354
+ "dtype": "torch.bfloat16",
2355
+ "sha256": "8c580f7575a955a4e20bf864e34486cfdaab448f97d72a9286fcf6b2da003bf6"
2356
+ },
2357
+ {
2358
+ "source_key": "model.visual.blocks.6.norm2.weight",
2359
+ "stored_key": "blocks.6.norm2.weight",
2360
+ "shape": [
2361
+ 1024
2362
+ ],
2363
+ "dtype": "torch.bfloat16",
2364
+ "sha256": "0e2fe864bd29a8be9f75e9d59e6177394b0b6b9af5ca69eeb26a91810725b659"
2365
+ },
2366
+ {
2367
+ "source_key": "model.visual.blocks.7.attn.proj.bias",
2368
+ "stored_key": "blocks.7.attn.proj.bias",
2369
+ "shape": [
2370
+ 1024
2371
+ ],
2372
+ "dtype": "torch.bfloat16",
2373
+ "sha256": "9cca012b5133c9c3264ecaeace4a5243c960d5f610ebc86e7405fa2696af663a"
2374
+ },
2375
+ {
2376
+ "source_key": "model.visual.blocks.7.attn.proj.weight",
2377
+ "stored_key": "blocks.7.attn.proj.weight",
2378
+ "shape": [
2379
+ 1024,
2380
+ 1024
2381
+ ],
2382
+ "dtype": "torch.bfloat16",
2383
+ "sha256": "e3faa107db4a78c2304caa047ad635e0a7a48670ebbdb634c50a23d5548a1520"
2384
+ },
2385
+ {
2386
+ "source_key": "model.visual.blocks.7.attn.qkv.bias",
2387
+ "stored_key": "blocks.7.attn.qkv.bias",
2388
+ "shape": [
2389
+ 3072
2390
+ ],
2391
+ "dtype": "torch.bfloat16",
2392
+ "sha256": "13b56b2d0f927bbe0ebc514db50690394f75cd1820f4ae04c300261adf05ad8f"
2393
+ },
2394
+ {
2395
+ "source_key": "model.visual.blocks.7.attn.qkv.weight",
2396
+ "stored_key": "blocks.7.attn.qkv.weight",
2397
+ "shape": [
2398
+ 3072,
2399
+ 1024
2400
+ ],
2401
+ "dtype": "torch.bfloat16",
2402
+ "sha256": "245e6747107d57e1552a7f4a95abf7fe40ffa0d08db990e37820d26a694abcda"
2403
+ },
2404
+ {
2405
+ "source_key": "model.visual.blocks.7.mlp.linear_fc1.bias",
2406
+ "stored_key": "blocks.7.mlp.linear_fc1.bias",
2407
+ "shape": [
2408
+ 4096
2409
+ ],
2410
+ "dtype": "torch.bfloat16",
2411
+ "sha256": "0d29ece3b7b813b562889a2b649ccc749fa0088c2edee4f8d48269980900a2ef"
2412
+ },
2413
+ {
2414
+ "source_key": "model.visual.blocks.7.mlp.linear_fc1.weight",
2415
+ "stored_key": "blocks.7.mlp.linear_fc1.weight",
2416
+ "shape": [
2417
+ 4096,
2418
+ 1024
2419
+ ],
2420
+ "dtype": "torch.bfloat16",
2421
+ "sha256": "e91ad20c42b9ba505462aee2bd595d9e2804db16d3af595afbec8bc447852476"
2422
+ },
2423
+ {
2424
+ "source_key": "model.visual.blocks.7.mlp.linear_fc2.bias",
2425
+ "stored_key": "blocks.7.mlp.linear_fc2.bias",
2426
+ "shape": [
2427
+ 1024
2428
+ ],
2429
+ "dtype": "torch.bfloat16",
2430
+ "sha256": "be0f1974bfd8220b5705eb03cc73656d5af80bd72d9ad314cad66010a94e2793"
2431
+ },
2432
+ {
2433
+ "source_key": "model.visual.blocks.7.mlp.linear_fc2.weight",
2434
+ "stored_key": "blocks.7.mlp.linear_fc2.weight",
2435
+ "shape": [
2436
+ 1024,
2437
+ 4096
2438
+ ],
2439
+ "dtype": "torch.bfloat16",
2440
+ "sha256": "613b645f34961897d789e856040bd72ef911b5f90d7c4f220edc63ae368aece4"
2441
+ },
2442
+ {
2443
+ "source_key": "model.visual.blocks.7.norm1.bias",
2444
+ "stored_key": "blocks.7.norm1.bias",
2445
+ "shape": [
2446
+ 1024
2447
+ ],
2448
+ "dtype": "torch.bfloat16",
2449
+ "sha256": "c3a5a56ebbf9fd793097dd662f87ea7b6ae6a509560f0f65fa5016ee87af87e6"
2450
+ },
2451
+ {
2452
+ "source_key": "model.visual.blocks.7.norm1.weight",
2453
+ "stored_key": "blocks.7.norm1.weight",
2454
+ "shape": [
2455
+ 1024
2456
+ ],
2457
+ "dtype": "torch.bfloat16",
2458
+ "sha256": "ae79ef0adfb422a71ec8b2e73f339c3f75d6f6654e6194937ec137e4fc3ff83a"
2459
+ },
2460
+ {
2461
+ "source_key": "model.visual.blocks.7.norm2.bias",
2462
+ "stored_key": "blocks.7.norm2.bias",
2463
+ "shape": [
2464
+ 1024
2465
+ ],
2466
+ "dtype": "torch.bfloat16",
2467
+ "sha256": "9d417bdf6bed94bee950892290a86ca970195dce24f7a4fdbb1e2a058d6c2cee"
2468
+ },
2469
+ {
2470
+ "source_key": "model.visual.blocks.7.norm2.weight",
2471
+ "stored_key": "blocks.7.norm2.weight",
2472
+ "shape": [
2473
+ 1024
2474
+ ],
2475
+ "dtype": "torch.bfloat16",
2476
+ "sha256": "0c98f7e075bb930791c41f0db436f91729787e2cfe983a041ef2d48b23416d5f"
2477
+ },
2478
+ {
2479
+ "source_key": "model.visual.blocks.8.attn.proj.bias",
2480
+ "stored_key": "blocks.8.attn.proj.bias",
2481
+ "shape": [
2482
+ 1024
2483
+ ],
2484
+ "dtype": "torch.bfloat16",
2485
+ "sha256": "d4059369dc9a761c283eedb8515347e7995b5f3fe457cf6dc5ec1c09514139c6"
2486
+ },
2487
+ {
2488
+ "source_key": "model.visual.blocks.8.attn.proj.weight",
2489
+ "stored_key": "blocks.8.attn.proj.weight",
2490
+ "shape": [
2491
+ 1024,
2492
+ 1024
2493
+ ],
2494
+ "dtype": "torch.bfloat16",
2495
+ "sha256": "cc15c3f0bb6e495e59f82810463454c5c4f109321ad2e5f24b9a87636e229884"
2496
+ },
2497
+ {
2498
+ "source_key": "model.visual.blocks.8.attn.qkv.bias",
2499
+ "stored_key": "blocks.8.attn.qkv.bias",
2500
+ "shape": [
2501
+ 3072
2502
+ ],
2503
+ "dtype": "torch.bfloat16",
2504
+ "sha256": "9caf035624521bc5399d4cc93cea2fa8746335a025a90c7215380905c618b93b"
2505
+ },
2506
+ {
2507
+ "source_key": "model.visual.blocks.8.attn.qkv.weight",
2508
+ "stored_key": "blocks.8.attn.qkv.weight",
2509
+ "shape": [
2510
+ 3072,
2511
+ 1024
2512
+ ],
2513
+ "dtype": "torch.bfloat16",
2514
+ "sha256": "bd0a00ef6d1c8ce4693d4da8fe740375cc7c1c29d7585649e33d902c6bd9e02b"
2515
+ },
2516
+ {
2517
+ "source_key": "model.visual.blocks.8.mlp.linear_fc1.bias",
2518
+ "stored_key": "blocks.8.mlp.linear_fc1.bias",
2519
+ "shape": [
2520
+ 4096
2521
+ ],
2522
+ "dtype": "torch.bfloat16",
2523
+ "sha256": "f1f82eb38f18794de48b6fe82c872c958dae900e634584e4af881b907dc4b529"
2524
+ },
2525
+ {
2526
+ "source_key": "model.visual.blocks.8.mlp.linear_fc1.weight",
2527
+ "stored_key": "blocks.8.mlp.linear_fc1.weight",
2528
+ "shape": [
2529
+ 4096,
2530
+ 1024
2531
+ ],
2532
+ "dtype": "torch.bfloat16",
2533
+ "sha256": "cb00bbeafc6d0cec9f0010a70d55e2b20ba551f2e054e6209b5be55be824c4d4"
2534
+ },
2535
+ {
2536
+ "source_key": "model.visual.blocks.8.mlp.linear_fc2.bias",
2537
+ "stored_key": "blocks.8.mlp.linear_fc2.bias",
2538
+ "shape": [
2539
+ 1024
2540
+ ],
2541
+ "dtype": "torch.bfloat16",
2542
+ "sha256": "c3700a25f91f42bb058ba12a683a1d530ce29c844faecc97486042d8b16b70e1"
2543
+ },
2544
+ {
2545
+ "source_key": "model.visual.blocks.8.mlp.linear_fc2.weight",
2546
+ "stored_key": "blocks.8.mlp.linear_fc2.weight",
2547
+ "shape": [
2548
+ 1024,
2549
+ 4096
2550
+ ],
2551
+ "dtype": "torch.bfloat16",
2552
+ "sha256": "1715953cd868e2a74c282d4c0e88939fc2b12550640173ba8705edfd8563448d"
2553
+ },
2554
+ {
2555
+ "source_key": "model.visual.blocks.8.norm1.bias",
2556
+ "stored_key": "blocks.8.norm1.bias",
2557
+ "shape": [
2558
+ 1024
2559
+ ],
2560
+ "dtype": "torch.bfloat16",
2561
+ "sha256": "7552227d1ddd71d5780e9ded54c96fec9d723e446126ccea760c65f31f7186d7"
2562
+ },
2563
+ {
2564
+ "source_key": "model.visual.blocks.8.norm1.weight",
2565
+ "stored_key": "blocks.8.norm1.weight",
2566
+ "shape": [
2567
+ 1024
2568
+ ],
2569
+ "dtype": "torch.bfloat16",
2570
+ "sha256": "3703a0f713f50d3de6ee20c70c1c13cc1b89904dbb8bcad8037bfc6f7df09e10"
2571
+ },
2572
+ {
2573
+ "source_key": "model.visual.blocks.8.norm2.bias",
2574
+ "stored_key": "blocks.8.norm2.bias",
2575
+ "shape": [
2576
+ 1024
2577
+ ],
2578
+ "dtype": "torch.bfloat16",
2579
+ "sha256": "9da81776e3a3b6b98ec668d6331b1b174f9c4c554461c62efffd4d3d8f5e3230"
2580
+ },
2581
+ {
2582
+ "source_key": "model.visual.blocks.8.norm2.weight",
2583
+ "stored_key": "blocks.8.norm2.weight",
2584
+ "shape": [
2585
+ 1024
2586
+ ],
2587
+ "dtype": "torch.bfloat16",
2588
+ "sha256": "60f27bf3028d886868ce407e5bd1b953f4bb57fefb035ca32b957562b3951618"
2589
+ },
2590
+ {
2591
+ "source_key": "model.visual.blocks.9.attn.proj.bias",
2592
+ "stored_key": "blocks.9.attn.proj.bias",
2593
+ "shape": [
2594
+ 1024
2595
+ ],
2596
+ "dtype": "torch.bfloat16",
2597
+ "sha256": "08ab989846e1050de1f0807e768c52825493fc24a08095a246f129d9293b14e7"
2598
+ },
2599
+ {
2600
+ "source_key": "model.visual.blocks.9.attn.proj.weight",
2601
+ "stored_key": "blocks.9.attn.proj.weight",
2602
+ "shape": [
2603
+ 1024,
2604
+ 1024
2605
+ ],
2606
+ "dtype": "torch.bfloat16",
2607
+ "sha256": "0d8b8516ecb8f6a8d81cce1f55325a5cde5d682c158d37e7a889733eb4c577fd"
2608
+ },
2609
+ {
2610
+ "source_key": "model.visual.blocks.9.attn.qkv.bias",
2611
+ "stored_key": "blocks.9.attn.qkv.bias",
2612
+ "shape": [
2613
+ 3072
2614
+ ],
2615
+ "dtype": "torch.bfloat16",
2616
+ "sha256": "fb12ad88fdf52e20f8d46bcede21322cbc1c6912d0047216a4c3bbb71687926c"
2617
+ },
2618
+ {
2619
+ "source_key": "model.visual.blocks.9.attn.qkv.weight",
2620
+ "stored_key": "blocks.9.attn.qkv.weight",
2621
+ "shape": [
2622
+ 3072,
2623
+ 1024
2624
+ ],
2625
+ "dtype": "torch.bfloat16",
2626
+ "sha256": "b457c7d033e417307a5b8bd4057cebc86b4b30f30b9704298341ef353ac5daa3"
2627
+ },
2628
+ {
2629
+ "source_key": "model.visual.blocks.9.mlp.linear_fc1.bias",
2630
+ "stored_key": "blocks.9.mlp.linear_fc1.bias",
2631
+ "shape": [
2632
+ 4096
2633
+ ],
2634
+ "dtype": "torch.bfloat16",
2635
+ "sha256": "6cc5ab8dd007fe94dd56ba1dbb20ddb7f8d0cb14e83e62f6cc5cf489e761f696"
2636
+ },
2637
+ {
2638
+ "source_key": "model.visual.blocks.9.mlp.linear_fc1.weight",
2639
+ "stored_key": "blocks.9.mlp.linear_fc1.weight",
2640
+ "shape": [
2641
+ 4096,
2642
+ 1024
2643
+ ],
2644
+ "dtype": "torch.bfloat16",
2645
+ "sha256": "9e9c3bbe0dada2b449f317cfb391f158e16b3f86bd750ee205bb1a91b25b5505"
2646
+ },
2647
+ {
2648
+ "source_key": "model.visual.blocks.9.mlp.linear_fc2.bias",
2649
+ "stored_key": "blocks.9.mlp.linear_fc2.bias",
2650
+ "shape": [
2651
+ 1024
2652
+ ],
2653
+ "dtype": "torch.bfloat16",
2654
+ "sha256": "ca01bc12788e04ae72878494cb06a6a43f5d31b7ac977b121628cb6be6b5ebf8"
2655
+ },
2656
+ {
2657
+ "source_key": "model.visual.blocks.9.mlp.linear_fc2.weight",
2658
+ "stored_key": "blocks.9.mlp.linear_fc2.weight",
2659
+ "shape": [
2660
+ 1024,
2661
+ 4096
2662
+ ],
2663
+ "dtype": "torch.bfloat16",
2664
+ "sha256": "9110b45920dccb3a4876414c9d84c489e526372dbc2414332e233e1f117d541d"
2665
+ },
2666
+ {
2667
+ "source_key": "model.visual.blocks.9.norm1.bias",
2668
+ "stored_key": "blocks.9.norm1.bias",
2669
+ "shape": [
2670
+ 1024
2671
+ ],
2672
+ "dtype": "torch.bfloat16",
2673
+ "sha256": "490553b048cde4248683d5bcfa8bd3f757c43992db6913e81db1b79d84b05e8e"
2674
+ },
2675
+ {
2676
+ "source_key": "model.visual.blocks.9.norm1.weight",
2677
+ "stored_key": "blocks.9.norm1.weight",
2678
+ "shape": [
2679
+ 1024
2680
+ ],
2681
+ "dtype": "torch.bfloat16",
2682
+ "sha256": "ed0e82ef5b3ec09ac4da1fa8e53d6b16aff2de7e119b6f3773062927753a8133"
2683
+ },
2684
+ {
2685
+ "source_key": "model.visual.blocks.9.norm2.bias",
2686
+ "stored_key": "blocks.9.norm2.bias",
2687
+ "shape": [
2688
+ 1024
2689
+ ],
2690
+ "dtype": "torch.bfloat16",
2691
+ "sha256": "e96a3880747aa9f42c73878ee71a927e631304f7508258a85c228c3dc6500cec"
2692
+ },
2693
+ {
2694
+ "source_key": "model.visual.blocks.9.norm2.weight",
2695
+ "stored_key": "blocks.9.norm2.weight",
2696
+ "shape": [
2697
+ 1024
2698
+ ],
2699
+ "dtype": "torch.bfloat16",
2700
+ "sha256": "8ed2c50e90f01fbfa26ff316da57af434a5b020473b1dcd920de42a803137a13"
2701
+ },
2702
+ {
2703
+ "source_key": "model.visual.merger.linear_fc1.bias",
2704
+ "stored_key": "merger.linear_fc1.bias",
2705
+ "shape": [
2706
+ 4096
2707
+ ],
2708
+ "dtype": "torch.bfloat16",
2709
+ "sha256": "d6658172d558ad239793f6a39558d1c967e7b0dd3e82f519ace2f65bd6c743ab"
2710
+ },
2711
+ {
2712
+ "source_key": "model.visual.merger.linear_fc1.weight",
2713
+ "stored_key": "merger.linear_fc1.weight",
2714
+ "shape": [
2715
+ 4096,
2716
+ 4096
2717
+ ],
2718
+ "dtype": "torch.bfloat16",
2719
+ "sha256": "f1888b11250b29ae3511f67f0414d778f43437f22000b5b90c2b9bf6c28e5952"
2720
+ },
2721
+ {
2722
+ "source_key": "model.visual.merger.linear_fc2.bias",
2723
+ "stored_key": "merger.linear_fc2.bias",
2724
+ "shape": [
2725
+ 2560
2726
+ ],
2727
+ "dtype": "torch.bfloat16",
2728
+ "sha256": "705f2e266ac98910d9172d9b335a7f7352df78cbdd6f8e637d15c54beadc9874"
2729
+ },
2730
+ {
2731
+ "source_key": "model.visual.merger.linear_fc2.weight",
2732
+ "stored_key": "merger.linear_fc2.weight",
2733
+ "shape": [
2734
+ 2560,
2735
+ 4096
2736
+ ],
2737
+ "dtype": "torch.bfloat16",
2738
+ "sha256": "2fa153b8fb28097ec5d28f9dd6ea84ec9ac8a4c5b5968d867d99e238b7b02799"
2739
+ },
2740
+ {
2741
+ "source_key": "model.visual.merger.norm.bias",
2742
+ "stored_key": "merger.norm.bias",
2743
+ "shape": [
2744
+ 1024
2745
+ ],
2746
+ "dtype": "torch.bfloat16",
2747
+ "sha256": "ff652f828b74639b8547e8421ccb7f25c5ea3107d748e43dac6cef20702a2b34"
2748
+ },
2749
+ {
2750
+ "source_key": "model.visual.merger.norm.weight",
2751
+ "stored_key": "merger.norm.weight",
2752
+ "shape": [
2753
+ 1024
2754
+ ],
2755
+ "dtype": "torch.bfloat16",
2756
+ "sha256": "16ebb133b30ae6750f69ca0d43032c82b0da0a39441679151970b5d254f7f33e"
2757
+ },
2758
+ {
2759
+ "source_key": "model.visual.patch_embed.proj.bias",
2760
+ "stored_key": "patch_embed.proj.bias",
2761
+ "shape": [
2762
+ 1024
2763
+ ],
2764
+ "dtype": "torch.bfloat16",
2765
+ "sha256": "a506a26a627cd872519f311d8de60e3b83c2898cdc8a2f5b9facef64fc160ca1"
2766
+ },
2767
+ {
2768
+ "source_key": "model.visual.patch_embed.proj.weight",
2769
+ "stored_key": "patch_embed.proj.weight",
2770
+ "shape": [
2771
+ 1024,
2772
+ 3,
2773
+ 2,
2774
+ 16,
2775
+ 16
2776
+ ],
2777
+ "dtype": "torch.bfloat16",
2778
+ "sha256": "5f99071446aded5f294c4d746edc457a47379982e064c39620543fed13ac175b"
2779
+ },
2780
+ {
2781
+ "source_key": "model.visual.pos_embed.weight",
2782
+ "stored_key": "pos_embed.weight",
2783
+ "shape": [
2784
+ 2304,
2785
+ 1024
2786
+ ],
2787
+ "dtype": "torch.bfloat16",
2788
+ "sha256": "9262980077f73fc984d234482f7b9c88f1091bddcdfa9eb2d34f6794ea69cebf"
2789
+ }
2790
+ ],
2791
+ "text_runtime": "0.1.3",
2792
+ "visual_finetuning": false,
2793
+ "base_commit": "not recorded in local source; identities pinned by hashes",
2794
+ "note": "Original base vision encoder+merger. Existing text backbone and pointer head stay unchanged. Not a standalone generative multimodal checkpoint."
2795
+ }
vision/example.py ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Run from any cwd: python /model/vision/example.py --model-dir /model --image image.png --request question.json"""
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+ from PIL import Image
6
+ from predictor import VisionDecisionEngine
7
+
8
+ def main():
9
+ p=argparse.ArgumentParser()
10
+ p.add_argument('--model-dir',required=True)
11
+ p.add_argument('--image',required=True)
12
+ p.add_argument('--request',required=True)
13
+ p.add_argument('--device',default='cuda')
14
+ a=p.parse_args()
15
+ engine=VisionDecisionEngine(a.model_dir,a.device)
16
+ with Image.open(a.image) as image:
17
+ result=engine.predict(json.loads(Path(a.request).read_text()),image)
18
+ print(json.dumps(result,ensure_ascii=False,indent=2))
19
+
20
+ if __name__=='__main__':main()
vision/example_request.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "NeoHorse-Jev-4B",
3
+ "state": "Look at the supplied image.",
4
+ "questions": {
5
+ "color": {
6
+ "type": "choice",
7
+ "instructions": "What is the main color of the image?",
8
+ "criteria": {"red": "Mostly red.", "blue": "Mostly blue.", "green": "Mostly green."}
9
+ }
10
+ }
11
+ }
vision/http_example.py ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Send one local image to either decision HTTP route using Python's standard library."""
2
+ import argparse
3
+ import base64
4
+ import json
5
+ import os
6
+ from pathlib import Path
7
+ from urllib.request import Request, urlopen
8
+
9
+ def main():
10
+ p=argparse.ArgumentParser()
11
+ p.add_argument('--image',required=True)
12
+ p.add_argument('--request')
13
+ p.add_argument('--base-url',default='http://127.0.0.1:8080')
14
+ p.add_argument('--endpoint',choices=['decision','systemone'],default='decision')
15
+ args=p.parse_args()
16
+ path=Path(args.image)
17
+ mime={'.png':'image/png','.jpg':'image/jpeg','.jpeg':'image/jpeg','.webp':'image/webp'}.get(path.suffix.lower())
18
+ if mime is None:p.error('Use PNG/JPEG/WebP')
19
+ if path.stat().st_size>4*1024*1024:p.error('Image file exceeds 4 MiB')
20
+ req=json.loads(Path(args.request).read_text()) if args.request else dict(
21
+ model='NeoHorse-Jev-4B',state='Look at the supplied image.',questions={
22
+ 'color':dict(type='choice',instructions='What is the dominant color?',criteria={'red':'red','blue':'blue','green':'green'})})
23
+ req.setdefault('model','NeoHorse-Jev-4B')
24
+ req['image']='data:'+mime+';base64,'+base64.b64encode(path.read_bytes()).decode()
25
+ headers={'Content-Type':'application/json'}
26
+ if os.environ.get('NEOHORSE_API_KEY'):headers['Authorization']='Bearer '+os.environ['NEOHORSE_API_KEY']
27
+ call=Request(args.base_url.rstrip('/')+'/v1/'+args.endpoint,
28
+ data=json.dumps(req,ensure_ascii=False).encode(),headers=headers,method='POST')
29
+ with urlopen(call,timeout=120) as response:
30
+ print(json.dumps(json.load(response),ensure_ascii=False,indent=2))
31
+
32
+ if __name__=='__main__':main()
vision/predictor.py ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ """Compatibility import for the bundled example; weights now live in backbone/."""
2
+ from neohorse_decision.vision import VisionDecisionEngine
vision/verification.json ADDED
@@ -0,0 +1,130 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "PASS",
3
+ "runtime": "0.2.0",
4
+ "vision_version": "unified-v2",
5
+ "visual_tensors": 297,
6
+ "vision_probability_max_abs_vs_split_release": 0.0,
7
+ "text_reference_requests": 16,
8
+ "text_wrapper_questions": 18,
9
+ "text_probability_max_abs": 0.0,
10
+ "wrapper_same_single_question_probability_max_abs": 0.0,
11
+ "split_question_vs_original_batched_probability_difference": 0.0041863322257995605,
12
+ "comparison_note": "Wrapper compared with text using identical single-question requests; original multi-question batches checked separately without splitting. BF16 split-vs-batch difference recorded, not treated as vision drift.",
13
+ "image_smoke_correct": 5,
14
+ "image_smoke_total": 5,
15
+ "rows": [
16
+ {
17
+ "task": "color",
18
+ "truth": "red",
19
+ "result": {
20
+ "model": "NeoHorse-Jev-4B",
21
+ "answers": {
22
+ "color": {
23
+ "type": "choice",
24
+ "choice": "red",
25
+ "probabilities": {
26
+ "red": 0.9999173879623413,
27
+ "blue": 4.41447009507101e-05,
28
+ "green": 3.853620364679955e-05
29
+ },
30
+ "confidence": 0.999876081943512
31
+ }
32
+ },
33
+ "input_tokens": 137,
34
+ "image_tokens": 96
35
+ },
36
+ "correct": true
37
+ },
38
+ {
39
+ "task": "color",
40
+ "truth": "blue",
41
+ "result": {
42
+ "model": "NeoHorse-Jev-4B",
43
+ "answers": {
44
+ "color": {
45
+ "type": "choice",
46
+ "choice": "blue",
47
+ "probabilities": {
48
+ "red": 2.3972206690814346e-05,
49
+ "blue": 0.9998923540115356,
50
+ "green": 8.369787246920168e-05
51
+ },
52
+ "confidence": 0.9998385310173035
53
+ }
54
+ },
55
+ "input_tokens": 137,
56
+ "image_tokens": 96
57
+ },
58
+ "correct": true
59
+ },
60
+ {
61
+ "task": "color",
62
+ "truth": "green",
63
+ "result": {
64
+ "model": "NeoHorse-Jev-4B",
65
+ "answers": {
66
+ "color": {
67
+ "type": "choice",
68
+ "choice": "green",
69
+ "probabilities": {
70
+ "red": 1.2651908946281765e-05,
71
+ "blue": 2.656863580341451e-05,
72
+ "green": 0.9999607801437378
73
+ },
74
+ "confidence": 0.9999411702156067
75
+ }
76
+ },
77
+ "input_tokens": 137,
78
+ "image_tokens": 96
79
+ },
80
+ "correct": true
81
+ },
82
+ {
83
+ "task": "position",
84
+ "truth": "left",
85
+ "result": {
86
+ "model": "NeoHorse-Jev-4B",
87
+ "answers": {
88
+ "side": {
89
+ "type": "choice",
90
+ "choice": "left",
91
+ "probabilities": {
92
+ "left": 0.9987630844116211,
93
+ "right": 0.0012369066243991256
94
+ },
95
+ "confidence": 0.9975261688232422
96
+ }
97
+ },
98
+ "input_tokens": 140,
99
+ "image_tokens": 96
100
+ },
101
+ "correct": true
102
+ },
103
+ {
104
+ "task": "position",
105
+ "truth": "right",
106
+ "result": {
107
+ "model": "NeoHorse-Jev-4B",
108
+ "answers": {
109
+ "side": {
110
+ "type": "choice",
111
+ "choice": "right",
112
+ "probabilities": {
113
+ "left": 0.00043360565905459225,
114
+ "right": 0.999566376209259
115
+ },
116
+ "confidence": 0.9991327524185181
117
+ }
118
+ },
119
+ "input_tokens": 140,
120
+ "image_tokens": 96
121
+ },
122
+ "correct": true
123
+ }
124
+ ],
125
+ "choice_noul_score_executed": true,
126
+ "stale_image_state_check": true,
127
+ "visual_finetuning": false,
128
+ "peak_cuda_allocated_bytes": 9825804800,
129
+ "scope": "Local-image connectivity and text parity, not multimodal benchmark validation"
130
+ }