Nex-N2-mini-GGUF / patch_gguf.py
Sebastian Aldrin
Add files using upload-large-folder tool
9f93a09 verified
Raw History Blame
3.57 kB
#!/usr/bin/env python3
"""Patch a fresh Nex-N2-mini GGUF to make it loadable by llama.cpp / Ollama / LM Studio.
The convert script (convert_hf_to_gguf.py) reads `mtp_num_hidden_layers: 1` from
config.json and writes `qwen35moe.nextn_predict_layers = 1` into the GGUF header.
It also writes `qwen35moe.block_count = 41` (40 transformer + 1 MTP). But the
public Nex-N2-mini safetensors don't include the MTP head weights, so llama.cpp's
loader refuses with:
missing tensor 'blk.39.nextn.eh_proj.weight'
This script flips two u32 metadata values in-place. Same length, no other bytes
shifted, takes ~30 seconds vs. re-converting from safetensors (~8 hours).
Usage:
python patch_gguf.py <path_to_gguf>
Safe to run on either the F16 GGUF (from convert) or an already-quantized GGUF.
Idempotent — running again is a no-op.
"""
from __future__ import annotations
import struct
import sys
from pathlib import Path
# GGUF KV value type sizes (for skipping)
SCALAR_SIZE = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
TYPE_STRING = 8
TYPE_ARRAY = 9
TYPE_U32 = 4
def skip_value(f, t):
if t in SCALAR_SIZE:
f.seek(SCALAR_SIZE[t], 1)
elif t == TYPE_STRING:
n, = struct.unpack('<Q', f.read(8))
f.seek(n, 1)
elif t == TYPE_ARRAY:
inner_t, = struct.unpack('<I', f.read(4))
cnt, = struct.unpack('<Q', f.read(8))
if inner_t in SCALAR_SIZE:
f.seek(SCALAR_SIZE[inner_t] * cnt, 1)
elif inner_t == TYPE_STRING:
for _ in range(cnt):
n, = struct.unpack('<Q', f.read(8))
f.seek(n, 1)
else:
raise ValueError(f"unsupported array inner type {inner_t}")
else:
raise ValueError(f"unsupported KV type {t}")
def read_str(f):
n, = struct.unpack('<Q', f.read(8))
return f.read(n).decode('utf-8', errors='replace')
def patch_u32(path: Path, key: str, new_value: int) -> bool:
"""Find a u32 KV pair by key name and overwrite its value. Returns True if patched."""
with open(path, 'r+b') as f:
assert f.read(4) == b'GGUF', "not a GGUF file"
f.read(4) # version
f.read(8) # tensor count
nkv, = struct.unpack('<Q', f.read(8))
for _ in range(nkv):
n, = struct.unpack('<Q', f.read(8))
k = f.read(n).decode('utf-8', errors='replace')
t, = struct.unpack('<I', f.read(4))
if k == key:
if t != TYPE_U32:
raise ValueError(f"{key} has type {t}, expected u32 (4)")
pos = f.tell()
old, = struct.unpack('<I', f.read(4))
if old == new_value:
print(f" {key} already = {new_value}, no change")
return False
f.seek(pos)
f.write(struct.pack('<I', new_value))
print(f" patched {key}: {old} -> {new_value}")
return True
skip_value(f, t)
raise ValueError(f"key not found: {key}")
def main() -> int:
if len(sys.argv) != 2:
print(__doc__)
return 1
path = Path(sys.argv[1]).resolve()
if not path.exists():
print(f"ERROR: {path} does not exist")
return 1
print(f"patching {path}")
patch_u32(path, "qwen35moe.nextn_predict_layers", 0)
patch_u32(path, "qwen35moe.block_count", 40)
print("done. The GGUF should now load in any recent llama.cpp / LM Studio / Ollama 0.12.6+")
return 0
if __name__ == "__main__":
sys.exit(main())