#!/usr/bin/env python3 """Patch a fresh Nex-N2-mini GGUF to make it loadable by llama.cpp / Ollama / LM Studio. The convert script (convert_hf_to_gguf.py) reads `mtp_num_hidden_layers: 1` from config.json and writes `qwen35moe.nextn_predict_layers = 1` into the GGUF header. It also writes `qwen35moe.block_count = 41` (40 transformer + 1 MTP). But the public Nex-N2-mini safetensors don't include the MTP head weights, so llama.cpp's loader refuses with: missing tensor 'blk.39.nextn.eh_proj.weight' This script flips two u32 metadata values in-place. Same length, no other bytes shifted, takes ~30 seconds vs. re-converting from safetensors (~8 hours). Usage: python patch_gguf.py Safe to run on either the F16 GGUF (from convert) or an already-quantized GGUF. Idempotent — running again is a no-op. """ from __future__ import annotations import struct import sys from pathlib import Path # GGUF KV value type sizes (for skipping) SCALAR_SIZE = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8} TYPE_STRING = 8 TYPE_ARRAY = 9 TYPE_U32 = 4 def skip_value(f, t): if t in SCALAR_SIZE: f.seek(SCALAR_SIZE[t], 1) elif t == TYPE_STRING: n, = struct.unpack(' bool: """Find a u32 KV pair by key name and overwrite its value. Returns True if patched.""" with open(path, 'r+b') as f: assert f.read(4) == b'GGUF', "not a GGUF file" f.read(4) # version f.read(8) # tensor count nkv, = struct.unpack(' {new_value}") return True skip_value(f, t) raise ValueError(f"key not found: {key}") def main() -> int: if len(sys.argv) != 2: print(__doc__) return 1 path = Path(sys.argv[1]).resolve() if not path.exists(): print(f"ERROR: {path} does not exist") return 1 print(f"patching {path}") patch_u32(path, "qwen35moe.nextn_predict_layers", 0) patch_u32(path, "qwen35moe.block_count", 40) print("done. The GGUF should now load in any recent llama.cpp / LM Studio / Ollama 0.12.6+") return 0 if __name__ == "__main__": sys.exit(main())