"""Approximate the verified mixed-Q8 edit as a small additive adapter. This is a new candidate, not an exact reconstruction of the BF16 surgery. Only exported-model behavioral evaluation may qualify it for publication. """ import json, math, time from pathlib import Path import numpy as np import torch from safetensors import safe_open from safetensors.torch import save_file from gguf import GGUFWriter R=Path('/workspace/bonsai2') if (R/'ADAPTERS_READY').exists():raise SystemExit(0) torch.manual_seed(1337) torch.set_num_threads(8) torch.backends.cuda.matmul.allow_tf32=False cfg=json.loads((R/'source_mlx/config.json').read_text()) src=safe_open(R/'source_mlx/model.safetensors',framework='pt',device='cpu') ref=safe_open(R/'reference/model.safetensors',framework='pt',device='cpu') ranks=(8,16,32) packs={rank:{} for rank in ranks} writers={} out=R/'adapters';out.mkdir(exist_ok=True) for rank in ranks: w=GGUFWriter(str(out/f'bonsai2-compact-r{rank}.gguf'),'qwen35') w.add_type('adapter');w.add_string('adapter.type','lora') w.add_float32('adapter.lora.alpha',float(rank)) w.add_name(f'Bonsai 2 Philadelphia compact rank {rank} correction') writers[rank]=w def decode(f,key,bits,group): p=f.get_tensor(key+'.weight').to(device='cuda',dtype=torch.int64) s=f.get_tensor(key+'.scales').cuda().float() b=f.get_tensor(key+'.biases').cuda().float() shifts=bits*torch.arange(32//bits,device='cuda',dtype=torch.int64) x=((p[...,None]>>shifts)&((1<