"""Prepare an auditable local release; this script never uploads anything.""" import json import shutil from pathlib import Path from safetensors import safe_open from common import ROOT, RUN, SOURCE, BASELINE, FAST, read_json, write_json, sha256, stamp from checkpoint import audit def verify_body(): import torch source = read_json(SOURCE / "model.safetensors.index.json")["weight_map"] target = read_json(FAST / "model.safetensors.index.json")["weight_map"] count = 0 for name, shard in source.items(): if name.startswith(("mtp.", "lm_head.", "model.language_model.embed_tokens.")): continue a, b = SOURCE / shard, FAST / target[name] if a.stat().st_ino != b.stat().st_ino or a.stat().st_dev != b.stat().st_dev: with safe_open(a,"pt") as left, safe_open(b,"pt") as right: assert torch.equal(left.get_tensor(name),right.get_tensor(name)), f"Unintended body change: {name}" count += 1 return count def main(): unchanged = verify_body() index = read_json(FAST / "model.safetensors.index.json") sizes = {"F64": 8, "F32": 4, "F16": 2, "BF16": 2, "I64": 8, "I32": 4, "I16": 2, "I8": 1, "U8": 1, "BOOL": 1} import math total = 0 for shard in set(index["weight_map"].values()): with safe_open(FAST / shard, "pt") as f: for name in f.keys(): tensor = f.get_slice(name) total += math.prod(tensor.get_shape()) * sizes[tensor.get_dtype()] index.setdefault("metadata", {})["total_size"] = total write_json(FAST / "model.safetensors.index.json", index) result = audit(FAST) provenance = FAST / "hyperqwen_provenance" provenance.mkdir(exist_ok=True) for name in ["LICENSE", "LICENSE-APACHE-2.0", "NOTICE"]: assert (SOURCE / name).exists(), f"Missing upstream {name}" shutil.copy2(SOURCE / name, FAST / name) for name in ["README.md", "QUANTIZATION_MANIFEST.json", "UPLOAD_MANIFEST.json", "recipe.yaml"]: if (SOURCE / name).exists(): shutil.copy2(SOURCE / name, provenance / ("upstream-" + name)) for name in ["manifest.json", "generation-manifest.json", "generation-summary.json", "vocabulary-report.json", "lm-head-report.json", "mtp-report.json", "mtp_hessians.pt.json"]: shutil.copy2(RUN / "calibration" / name, provenance / name) for source in list((ROOT/"swift15").glob("*.py")) + [ROOT/"swift15/README.md", ROOT/"drafter/gptq_utils.py", ROOT/"drafter/capture.py", ROOT/"drafter/train_mtp.py", ROOT/"prepare/build_draft_vocab.py", ROOT/"prepare/quant_heads_stream.py"]: destination = provenance/"recipe"/source.relative_to(ROOT) destination.parent.mkdir(parents=True,exist_ok=True) shutil.copy2(source,destination) for name in ["runtime-compatibility.json", "preflight-validation.json"]: if (RUN / name).exists(): shutil.copy2(RUN / name, provenance / name) mtp_replay = read_json(RUN / "calibration/mtp_hessians.pt.json") recipe = {"created": stamp(), "source": read_json(RUN / "source.json"), "variant": "fast", "preflight_only": "preflight" in RUN.parts, "unchanged_body_tensors_verified": unchanged, "body": "unchanged asymmetric AWQ INT4 group128", "embedding_bits": 8, "lm_head_bits": 4, "mtp_bits": 4, "group_size": 128, "act_order": False, "gptq": {"damping": .01, "lm_head_rows": read_json(RUN / "calibration/lm-head-report.json")["calibration_rows"], "mtp_examples": len(mtp_replay["examples"]), "mtp_depths": mtp_replay["depths"]}, "pristine_inputs": "Heads quantized directly from source BF16 tensors, not from INT8 tensors", "calibration": read_json(RUN / "calibration/manifest.json"), "runtime": read_json(RUN / "environment.json"), "script_hashes": {str(p.relative_to(ROOT)): sha256(p) for p in list((ROOT / "swift15").glob("*.py")) + [ROOT / "drafter/gptq_utils.py", ROOT / "drafter/capture.py", ROOT / "drafter/train_mtp.py", ROOT / "prepare/build_draft_vocab.py"]}} write_json(FAST / "hyperqwen-build.json", recipe) write_json(FAST / "QUANTIZATION_MANIFEST.json", recipe) # The inherited upload list describes the upstream checkpoint, not this one. if (FAST / "UPLOAD_MANIFEST.json").exists(): (FAST / "UPLOAD_MANIFEST.json").unlink() with (FAST / "NOTICE").open("a") as out: out.write("\nHyperQwen-compatible local derivative: INT8 embeddings; calibrated GPTQ INT4 output and MTP linear weights; reduced MTP draft vocabulary. The upstream AWQ body is unchanged. Quantization recipe and source attribution are included in hyperqwen-build.json.\n") report = RUN / "REPORT.md" measured = report.read_text() if report.exists() else "Evaluation is pending. No throughput or quality claim has been established for this derivative." card = """--- license: other license_name: swift-open-license-1.0 license_link: LICENSE base_model: ukisai/Swift-1.5-Qwen3.8-27b base_model_relation: quantized library_name: vllm pipeline_tag: text-generation tags: - compressed-tensors - awq - gptq - hyperqwen --- # Swift 1.5 Qwen3.8 27B — HyperQwen fast derivative Local release candidate. This is an independently prepared quantization of UkisAI's Swift 1.5, not an official UkisAI or HyperQwen release. The original asymmetric AWQ INT4 body is unchanged. Embeddings use INT8 group128; the full output head and eight MTP linear matrices use calibrated GPTQ INT4 group128. GPTQ starts from the upstream BF16 head tensors. The MTP draft head uses a subset of the output head selected using fresh Swift responses. This subset only limits draft proposals; the target retains its full vocabulary. This is not fine-tuning. Requires the patched HyperQwen runtime, including INT8 embeddings and reduced MTP vocabulary support. The tested package versions and recipe are in hyperqwen-build.json. Do not assume an unpatched Transformers/vLLM installation can load this checkpoint. Multi-user serving can use the W4A16 body without MTP. Optional INT8 activations are a separate runtime setting and require asymmetric Marlin support; they are not a property of the stored checkpoint. The upstream vision tensors are retained, but the initial evaluation is text-only. Calibration sources, pinned revisions and whole-example holdout separation are documented under hyperqwen_provenance. Raw benchmark answers and calibration examples are not included. See LICENSE, LICENSE-APACHE-2.0 and NOTICE for the upstream license and attribution. See hyperqwen_provenance/upstream-README.md for the original model information. ## Local evaluation """ + measured + "\n" if recipe["preflight_only"]: card = card.replace("Local release candidate.", "Integration-test checkpoint only; exclude this small-subset preflight model from publication and final comparisons.") (FAST / "README.md").write_text(card) files = sorted(p.relative_to(FAST).as_posix() for p in FAST.rglob("*") if p.is_file() and ".bak" not in p.name and not p.name.endswith(".tmp") and ".cache" not in p.parts) write_json(RUN / "release-files.json", files) print("Prepared local release:", FAST, ";", len(files), "files; no upload performed") if __name__ == "__main__": main()