Eliovp
Publish balanced MXFP4 conversion of abenzerps BF16 fine-tune
f5b159e
Raw History Blame Contribute Delete
9.18 kB
"""Losslessly extract the pinned BF16 GGUF; audit against the pinned upstream.
CPU-only preparation. This deliberately accepts only this BF16 checkpoint,
not arbitrary quantized GGUFs. No executable code is loaded from the model.
"""
import argparse
import hashlib
import json
import math
import os
from pathlib import Path
import shutil
import struct
import numpy as np
parser=argparse.ArgumentParser(description=__doc__)
parser.add_argument('--gguf',type=Path,required=True)
parser.add_argument('--original-snapshot',type=Path,required=True)
parser.add_argument('--output-dir',type=Path,required=True)
args=parser.parse_args()
ROOT=args.output_dir.resolve()
ROOT.mkdir(parents=True,exist_ok=False)
ORIGINAL=args.original_snapshot.resolve()
ORIGINAL_REVISION='790c92633540aa0cb11d9abf19eb46d861714758'
REVISION = '40319fb15542f0ad22921e0124a191a8a935a60a'
EXPECTED_HASH = 'f151c683a8aed4b310777017ebbbe3f2180f1180f7867115171adb7d50b0762a'
SOURCE = args.gguf.resolve()
SNAPSHOT = ROOT / 'bf16-snapshot'
def digest(path):
with Path(path).open('rb') as f:
return hashlib.file_digest(f, 'sha256').hexdigest()
def save_json(path, value):
Path(path).write_text(json.dumps(value, indent=2) + '\n')
part = SOURCE.with_suffix('.gguf.part')
source = SOURCE if SOURCE.exists() else part
assert source.stat().st_size == 14230272800
actual = digest(source)
assert actual == EXPECTED_HASH, actual
if source == part:
part.rename(SOURCE)
print('SOURCE SHA256 VERIFIED', actual, flush=True)
with SOURCE.open('rb') as f:
def scalar(fmt):
fmt = '<' + fmt
return struct.unpack(fmt, f.read(struct.calcsize(fmt)))[0]
def string():
n = scalar('Q')
assert n <= 1048576
return f.read(n).decode('utf-8')
def value(kind):
if kind == 8:
return string()
return scalar({0:'B', 1:'b', 2:'H', 3:'h', 4:'I', 5:'i', 6:'f', 7:'?', 10:'Q', 11:'q', 12:'d'}[kind])
assert f.read(4) == b'GGUF'
assert scalar('I') == 3
count, kv_count = scalar('Q'), scalar('Q')
assert count == 297 and kv_count < 1000
metadata = {}
for _ in range(kv_count):
key = string()
assert key not in metadata
metadata[key] = value(scalar('I'))
assert metadata['general.architecture'] == 'qwen_image21'
tensors = []
names = set()
for _ in range(count):
name, ndims = string(), scalar('I')
assert name not in names and 1 <= ndims <= 4
names.add(name)
shape = list(reversed([scalar('Q') for _ in range(ndims)]))
assert all(0 < d <= 1048576 for d in shape)
dtype, offset = scalar('I'), scalar('Q')
assert dtype == 30, (name, dtype)
tensors.append(dict(name=name, shape=shape, offset=offset, bytes=math.prod(shape)*2))
header_end = f.tell()
alignment = metadata.get('general.alignment', 32)
assert alignment > 0 and alignment <= 4096 and alignment & (alignment-1) == 0
data_start = (header_end + alignment-1) // alignment * alignment
end = 0
for t in sorted(tensors, key=lambda t: t['offset']):
assert t['offset'] % alignment == 0 and t['offset'] >= end
end = t['offset'] + t['bytes']
assert data_start + end <= SOURCE.stat().st_size
assert data_start + end == SOURCE.stat().st_size
# The author retained Diffusers parameter names. Verify that mapping exactly.
original = {}
for path in sorted((ORIGINAL/'transformer').glob('*.safetensors')):
with path.open('rb') as f:
header_size = struct.unpack('<Q', f.read(8))[0]
header = json.loads(f.read(header_size))
for name, row in header.items():
if name == '__metadata__':
continue
assert name not in original
original[name] = dict(row, path=path, start=8+header_size+row['data_offsets'][0])
assert names == set(original), (sorted(names-set(original)), sorted(set(original)-names))
for t in tensors:
row = original[t['name']]
assert t['shape'] == row['shape'] and row['dtype'] == 'BF16', (t, row)
SNAPSHOT.mkdir(exist_ok=False)
(SNAPSHOT/'transformer').mkdir()
for name in ('text_encoder', 'vae', 'processor', 'scheduler'):
os.symlink(ORIGINAL/name, SNAPSHOT/name, target_is_directory=True)
for name in ('LICENSE', 'model_index.json'):
if (ORIGINAL/name).exists():
shutil.copyfile(ORIGINAL/name, SNAPSHOT/name)
shutil.copyfile(ORIGINAL/'transformer/config.json', SNAPSHOT/'transformer/config.json')
output = SNAPSHOT/'transformer/diffusion_pytorch_model.safetensors'
notice = 'BF16 checkpoint published by abenzerps, losslessly extracted from GGUF by EliovpAI. Built with Qwen. Training recipe unverified.'
header = {'__metadata__': {'format': 'pt', 'modification_notice': notice}}
offset = 0
for t in tensors:
header[t['name']] = dict(dtype='BF16', shape=t['shape'], data_offsets=[offset, offset+t['bytes']])
offset += t['bytes']
encoded = json.dumps(header, separators=(',', ':')).encode()
encoded += b' ' * (-len(encoded) % 8)
audit = dict(status='running', source_repository='abenzerps/Qwen-Image-2.1-Uncensored-GGUF',
source_revision=REVISION, source_sha256=actual, source_bytes=SOURCE.stat().st_size,
gguf_metadata=metadata, extraction='BF16 raw-byte copy, no numeric conversion',
tensors=[], changed_tensors=0, changed_elements=0, total_elements=0,
original_revision=ORIGINAL_REVISION, source_data_start=data_start)
with SOURCE.open('rb') as src, output.open('xb') as dest:
dest.write(struct.pack('<Q', len(encoded)))
dest.write(encoded)
for index, t in enumerate(tensors):
src.seek(data_start+t['offset'])
sha = hashlib.sha256()
remaining = t['bytes']
changed = 0
sq_diff = sq_ref = max_abs = 0.0
orig = original[t['name']]
with orig['path'].open('rb') as ref:
ref.seek(orig['start'])
while remaining:
n = min(8*1024*1024, remaining)
chunk = src.read(n)
ref_chunk = ref.read(n)
assert len(chunk) == len(ref_chunk) == n
a, b = np.frombuffer(chunk, '<u2'), np.frombuffer(ref_chunk, '<u2')
af = (a.astype(np.uint32) << 16).view(np.float32)
bf = (b.astype(np.uint32) << 16).view(np.float32)
assert np.isfinite(af).all() and np.isfinite(bf).all(), t['name']
delta = af.astype(np.float64) - bf
changed += int(np.count_nonzero(a != b))
sq_diff += float(np.sum(delta*delta))
sq_ref += float(np.sum(bf.astype(np.float64)**2))
max_abs = max(max_abs, float(np.max(np.abs(delta))))
dest.write(chunk)
sha.update(chunk)
remaining -= n
record = dict(t, sha256=sha.hexdigest(), changed_elements_vs_original=changed,
relative_l2_vs_original=math.sqrt(sq_diff/max(sq_ref, 1e-300)), max_abs_vs_original=max_abs)
audit['tensors'].append(record)
audit['changed_tensors'] += int(changed > 0)
audit['changed_elements'] += changed
audit['total_elements'] += t['bytes']//2
if index % 25 == 0:
print('EXTRACTED', index+1, '/', count, t['name'], 'changed elements:', changed, flush=True)
# Independently read the completed safetensors and verify every raw tensor hash.
with output.open('rb') as f:
n = struct.unpack('<Q', f.read(8))[0]
complete_header = json.loads(f.read(n))
for t in audit['tensors']:
row = complete_header[t['name']]
f.seek(8+n+row['data_offsets'][0])
sha = hashlib.sha256()
remaining = t['bytes']
while remaining:
chunk = f.read(min(8*1024*1024, remaining))
assert chunk
sha.update(chunk)
remaining -= len(chunk)
assert sha.hexdigest() == t['sha256'], t['name']
audit['output_sha256'] = digest(output)
audit['output_bytes'] = output.stat().st_size
audit['status'] = 'verified_lossless_bf16_extraction'
save_json(ROOT/'source-audit.json', audit)
assert audit['changed_tensors'] > 0, 'Source is identical to original; do not market this as a verified distinct fine-tune.'
# Retain the pinned original companion hashes; update only the transformer pin.
pin = json.loads((Path(__file__).resolve().parent/'original-source-manifest.json').read_text())
pin.update(repository=audit['source_repository'], revision=REVISION,
original_companions_repository='Qwen/Qwen-Image-2.1', original_companions_revision=ORIGINAL_REVISION,
bf16_gguf=dict(file=SOURCE.name, bytes=SOURCE.stat().st_size, sha256=actual),
extraction_audit_sha256=digest(ROOT/'source-audit.json'))
pin['files'] = [r for r in pin['files'] if not (r['path'].startswith('transformer/') and r['path'].endswith('.safetensors'))]
pin['files'].append(dict(path='transformer/'+output.name, bytes=output.stat().st_size, sha256=audit['output_sha256']))
save_json(ROOT/'source-manifest.json', pin)
print('COMPLETE', audit['changed_tensors'], 'changed tensors;', audit['changed_elements'], 'changed elements of', audit['total_elements'], flush=True)