Sebastian Aldrin commited on
Commit
9f93a09
·
verified ·
1 Parent(s): 11e0c35

Add files using upload-large-folder tool

Browse files
Files changed (6) hide show
  1. .gitattributes +2 -0
  2. Modelfile +32 -0
  3. Nex-N2-mini-IQ3_XXS.gguf +3 -0
  4. README.md +88 -0
  5. imatrix.dat +3 -0
  6. patch_gguf.py +104 -0
.gitattributes CHANGED
@@ -33,3 +33,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ imatrix.dat filter=lfs diff=lfs merge=lfs -text
37
+ Nex-N2-mini-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Modelfile ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Ollama Modelfile for Nex-N2-mini-IQ3_XXS.
2
+ # Requires Ollama 0.19 or newer (qwen35moe arch landed in llama.cpp PR #19468 on 2026-02-10).
3
+ #
4
+ # Usage:
5
+ # ollama create nex-n2-mini -f Modelfile
6
+ # ollama run nex-n2-mini "your prompt"
7
+
8
+ FROM ./Nex-N2-mini-IQ3_XXS.gguf
9
+
10
+ # Context window (4096 default; bump up to 262144 if you have GPU memory)
11
+ PARAMETER num_ctx 4096
12
+
13
+ # Offload everything to GPU. Drop if you have less than ~14 GB GPU memory.
14
+ PARAMETER num_gpu 999
15
+
16
+ # Leave 2 cores for the desktop on a typical 8-core laptop.
17
+ PARAMETER num_thread 6
18
+
19
+ # Single-user laptop: don't allocate multiple parallel KV caches.
20
+ PARAMETER num_predict -1
21
+
22
+ # Pin the model in RAM after first load (avoids re-mmap on each request).
23
+ PARAMETER use_mlock true
24
+
25
+ # Sampling defaults that match the model's training (Qwen 3.5 family).
26
+ PARAMETER temperature 0.4
27
+ PARAMETER top_p 0.95
28
+ PARAMETER top_k 64
29
+
30
+ # ChatML stop tokens (Nex-N2-mini uses Qwen 3.5 chat format).
31
+ PARAMETER stop "<|im_end|>"
32
+ PARAMETER stop "<|im_start|>"
Nex-N2-mini-IQ3_XXS.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb18917549bf87c3ce6fc8294646c7f83cf333759782ecce89fde16fff971d66
3
+ size 13623758848
README.md CHANGED
@@ -1,3 +1,91 @@
1
  ---
2
  license: apache-2.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
  license: apache-2.0
3
+ base_model: nex-agi/Nex-N2-mini
4
+ base_model_relation: quantized
5
+ library_name: gguf
6
+ pipeline_tag: text-generation
7
+ tags:
8
+ - gguf
9
+ - quantized
10
+ - imatrix
11
+ - iq3_xxs
12
+ - qwen3
13
+ - qwen35moe
14
+ - moe
15
+ - 35b
16
+ - 3b-active
17
+ - agentic
18
+ - tool-use
19
+ - reasoning
20
+ language:
21
+ - en
22
  ---
23
+
24
+ # Nex-N2-mini IQ3_XXS
25
+
26
+ imatrix-calibrated IQ3_XXS of nex-n2-mini. ~13gb, fits in 15gb GPU memory with room left for context. mostly made it because i wanted to run this model on a laptop iGPU and the existing quants were all 16gb+.
27
+
28
+ gets ~14 tok/s on CPU only on a ryzen 7 pro 7735u, more with vulkan offload.
29
+
30
+ ## you'll hit this if you quantize it yourself
31
+
32
+ ```
33
+ missing tensor 'blk.39.nextn.eh_proj.weight'
34
+ ```
35
+
36
+ took me forever to figure out. nex agi didnt release the MTP weights with the model, but the config says they're there, so the convert script writes "has MTP" into the GGUF header and then nothing loads it.
37
+
38
+ two settings to patch:
39
+
40
+ ```
41
+ qwen35moe.nextn_predict_layers: 1 → 0
42
+ qwen35moe.block_count: 41 → 40
43
+ ```
44
+
45
+ `patch_gguf.py` in the repo does it. 4-byte edits, doesnt shift anything else, takes 30s. way faster than re-converting from safetensors (8h).
46
+
47
+ ## using it
48
+
49
+ needs a llama.cpp from after 2026-02-10 (when qwen35moe arch landed in PR #19468).
50
+
51
+ LM Studio: drop the gguf in `~/.lmstudio/models/<you>/Nex-N2-mini-GGUF/`, load it. update the bundled llama.cpp runtime to 2.13+ if loading fails.
52
+
53
+ Ollama 0.19+:
54
+ ```
55
+ ollama create nex-n2-mini -f Modelfile
56
+ ollama run nex-n2-mini "hi"
57
+ ```
58
+
59
+ llama-cli (ChatML):
60
+ ```
61
+ llama-cli -m Nex-N2-mini-IQ3_XXS.gguf -ngl 999 \
62
+ -p $'<|im_start|>system\nyou are helpful<|im_end|>\n<|im_start|>user\nhi<|im_end|>\n<|im_start|>assistant\n' \
63
+ -n 100
64
+ ```
65
+
66
+ ollama before 0.19 wont work — too old to know the qwen35moe arch.
67
+
68
+ ## stuff to know
69
+
70
+ - its a reasoning model so outputs have `<think>...</think>` blocks. handle that or strip it
71
+ - no MTP speedup since the weights arent in the public release
72
+ - text only. the base model's config.json has vision/video token slots but llama.cpp's qwen35moe converter is text-only (PR #19468 title literally says "no vision"). dont expect image support
73
+ - Q3 means maybe 1-3% benchmark drop vs Q4. for chat/tool use i cant tell
74
+
75
+ ## how i made it
76
+
77
+ ```
78
+ huggingface-cli download nex-agi/Nex-N2-mini --local-dir source
79
+ python convert_hf_to_gguf.py source --outtype f16
80
+ python patch_gguf.py source/*-F16.gguf # whatever the convert script named it
81
+ llama-imatrix -m <f16.gguf> -f calibration.txt -o imatrix.dat --chunks 50
82
+ llama-quantize --imatrix imatrix.dat <f16.gguf> out.gguf IQ3_XXS
83
+ ```
84
+
85
+ calibration text was Pride and Prejudice from gutenberg plus the nex-n2 README. imatrix-aware quantizer kept attention tensors at Q4_K and pushed expert FFN weights down to IQ3, ended up at 3.14 bpw avg.
86
+
87
+ `imatrix.dat` is in the repo if you wanna re-quantize to something else.
88
+
89
+ ---
90
+
91
+ credit to nex agi for the model, Qwen team for the base arch, ggml-org for llama.cpp. apache 2.0, same as the base.
imatrix.dat ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ffe3acb6b3ee36c7ae5c554a6b44c0d5645c126e39e37605ed3afa56484bb6f0
3
+ size 192223904
patch_gguf.py ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Patch a fresh Nex-N2-mini GGUF to make it loadable by llama.cpp / Ollama / LM Studio.
3
+
4
+ The convert script (convert_hf_to_gguf.py) reads `mtp_num_hidden_layers: 1` from
5
+ config.json and writes `qwen35moe.nextn_predict_layers = 1` into the GGUF header.
6
+ It also writes `qwen35moe.block_count = 41` (40 transformer + 1 MTP). But the
7
+ public Nex-N2-mini safetensors don't include the MTP head weights, so llama.cpp's
8
+ loader refuses with:
9
+
10
+ missing tensor 'blk.39.nextn.eh_proj.weight'
11
+
12
+ This script flips two u32 metadata values in-place. Same length, no other bytes
13
+ shifted, takes ~30 seconds vs. re-converting from safetensors (~8 hours).
14
+
15
+ Usage:
16
+ python patch_gguf.py <path_to_gguf>
17
+
18
+ Safe to run on either the F16 GGUF (from convert) or an already-quantized GGUF.
19
+ Idempotent — running again is a no-op.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import struct
24
+ import sys
25
+ from pathlib import Path
26
+
27
+ # GGUF KV value type sizes (for skipping)
28
+ SCALAR_SIZE = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
29
+ TYPE_STRING = 8
30
+ TYPE_ARRAY = 9
31
+ TYPE_U32 = 4
32
+
33
+
34
+ def skip_value(f, t):
35
+ if t in SCALAR_SIZE:
36
+ f.seek(SCALAR_SIZE[t], 1)
37
+ elif t == TYPE_STRING:
38
+ n, = struct.unpack('<Q', f.read(8))
39
+ f.seek(n, 1)
40
+ elif t == TYPE_ARRAY:
41
+ inner_t, = struct.unpack('<I', f.read(4))
42
+ cnt, = struct.unpack('<Q', f.read(8))
43
+ if inner_t in SCALAR_SIZE:
44
+ f.seek(SCALAR_SIZE[inner_t] * cnt, 1)
45
+ elif inner_t == TYPE_STRING:
46
+ for _ in range(cnt):
47
+ n, = struct.unpack('<Q', f.read(8))
48
+ f.seek(n, 1)
49
+ else:
50
+ raise ValueError(f"unsupported array inner type {inner_t}")
51
+ else:
52
+ raise ValueError(f"unsupported KV type {t}")
53
+
54
+
55
+ def read_str(f):
56
+ n, = struct.unpack('<Q', f.read(8))
57
+ return f.read(n).decode('utf-8', errors='replace')
58
+
59
+
60
+ def patch_u32(path: Path, key: str, new_value: int) -> bool:
61
+ """Find a u32 KV pair by key name and overwrite its value. Returns True if patched."""
62
+ with open(path, 'r+b') as f:
63
+ assert f.read(4) == b'GGUF', "not a GGUF file"
64
+ f.read(4) # version
65
+ f.read(8) # tensor count
66
+ nkv, = struct.unpack('<Q', f.read(8))
67
+ for _ in range(nkv):
68
+ n, = struct.unpack('<Q', f.read(8))
69
+ k = f.read(n).decode('utf-8', errors='replace')
70
+ t, = struct.unpack('<I', f.read(4))
71
+ if k == key:
72
+ if t != TYPE_U32:
73
+ raise ValueError(f"{key} has type {t}, expected u32 (4)")
74
+ pos = f.tell()
75
+ old, = struct.unpack('<I', f.read(4))
76
+ if old == new_value:
77
+ print(f" {key} already = {new_value}, no change")
78
+ return False
79
+ f.seek(pos)
80
+ f.write(struct.pack('<I', new_value))
81
+ print(f" patched {key}: {old} -> {new_value}")
82
+ return True
83
+ skip_value(f, t)
84
+ raise ValueError(f"key not found: {key}")
85
+
86
+
87
+ def main() -> int:
88
+ if len(sys.argv) != 2:
89
+ print(__doc__)
90
+ return 1
91
+ path = Path(sys.argv[1]).resolve()
92
+ if not path.exists():
93
+ print(f"ERROR: {path} does not exist")
94
+ return 1
95
+
96
+ print(f"patching {path}")
97
+ patch_u32(path, "qwen35moe.nextn_predict_layers", 0)
98
+ patch_u32(path, "qwen35moe.block_count", 40)
99
+ print("done. The GGUF should now load in any recent llama.cpp / LM Studio / Ollama 0.12.6+")
100
+ return 0
101
+
102
+
103
+ if __name__ == "__main__":
104
+ sys.exit(main())