File size: 3,565 Bytes
9f93a09
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
#!/usr/bin/env python3
"""Patch a fresh Nex-N2-mini GGUF to make it loadable by llama.cpp / Ollama / LM Studio.

The convert script (convert_hf_to_gguf.py) reads `mtp_num_hidden_layers: 1` from
config.json and writes `qwen35moe.nextn_predict_layers = 1` into the GGUF header.
It also writes `qwen35moe.block_count = 41` (40 transformer + 1 MTP). But the
public Nex-N2-mini safetensors don't include the MTP head weights, so llama.cpp's
loader refuses with:

    missing tensor 'blk.39.nextn.eh_proj.weight'

This script flips two u32 metadata values in-place. Same length, no other bytes
shifted, takes ~30 seconds vs. re-converting from safetensors (~8 hours).

Usage:
    python patch_gguf.py <path_to_gguf>

Safe to run on either the F16 GGUF (from convert) or an already-quantized GGUF.
Idempotent — running again is a no-op.
"""
from __future__ import annotations

import struct
import sys
from pathlib import Path

# GGUF KV value type sizes (for skipping)
SCALAR_SIZE = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
TYPE_STRING = 8
TYPE_ARRAY = 9
TYPE_U32 = 4


def skip_value(f, t):
    if t in SCALAR_SIZE:
        f.seek(SCALAR_SIZE[t], 1)
    elif t == TYPE_STRING:
        n, = struct.unpack('<Q', f.read(8))
        f.seek(n, 1)
    elif t == TYPE_ARRAY:
        inner_t, = struct.unpack('<I', f.read(4))
        cnt, = struct.unpack('<Q', f.read(8))
        if inner_t in SCALAR_SIZE:
            f.seek(SCALAR_SIZE[inner_t] * cnt, 1)
        elif inner_t == TYPE_STRING:
            for _ in range(cnt):
                n, = struct.unpack('<Q', f.read(8))
                f.seek(n, 1)
        else:
            raise ValueError(f"unsupported array inner type {inner_t}")
    else:
        raise ValueError(f"unsupported KV type {t}")


def read_str(f):
    n, = struct.unpack('<Q', f.read(8))
    return f.read(n).decode('utf-8', errors='replace')


def patch_u32(path: Path, key: str, new_value: int) -> bool:
    """Find a u32 KV pair by key name and overwrite its value. Returns True if patched."""
    with open(path, 'r+b') as f:
        assert f.read(4) == b'GGUF', "not a GGUF file"
        f.read(4)         # version
        f.read(8)         # tensor count
        nkv, = struct.unpack('<Q', f.read(8))
        for _ in range(nkv):
            n, = struct.unpack('<Q', f.read(8))
            k = f.read(n).decode('utf-8', errors='replace')
            t, = struct.unpack('<I', f.read(4))
            if k == key:
                if t != TYPE_U32:
                    raise ValueError(f"{key} has type {t}, expected u32 (4)")
                pos = f.tell()
                old, = struct.unpack('<I', f.read(4))
                if old == new_value:
                    print(f"  {key} already = {new_value}, no change")
                    return False
                f.seek(pos)
                f.write(struct.pack('<I', new_value))
                print(f"  patched {key}: {old} -> {new_value}")
                return True
            skip_value(f, t)
    raise ValueError(f"key not found: {key}")


def main() -> int:
    if len(sys.argv) != 2:
        print(__doc__)
        return 1
    path = Path(sys.argv[1]).resolve()
    if not path.exists():
        print(f"ERROR: {path} does not exist")
        return 1

    print(f"patching {path}")
    patch_u32(path, "qwen35moe.nextn_predict_layers", 0)
    patch_u32(path, "qwen35moe.block_count", 40)
    print("done. The GGUF should now load in any recent llama.cpp / LM Studio / Ollama 0.12.6+")
    return 0


if __name__ == "__main__":
    sys.exit(main())