Dragoy's picture
Swift 1.5 huihui-style abliterated NVFP4/FP8 NInfer v3 (text, vision, MTP, DFlash2)
fe202c6 verified
Raw History Blame Contribute Delete
3.04 kB
#!/usr/bin/env python3
"""NVFP4+FP8 quantization for the NInfer qwen3.8-27b/nvfp4 profile.
Replicates the allocation published in unsloth/Qwen3.8-27B-NVFP4's config.json — the canonical
quantized source that tools/convert/qwen3_8_27b/convert_nvfp4.py consumes. The recipe is read
from a JSON file rather than reconstructed, so it cannot drift from upstream by paraphrase.
Usage: quantize_nvfp4.py <bf16_src> <out_dir> [--recipe unsloth_qconfig.json]
[--samples 32] [--seq 2048]
CALIBRATION SIZE MATTERS ENORMOUSLY. The only data-dependent value in this recipe is one
`input_global_scale` scalar per NVFP4 matrix (static_minmax); the FP8 half is fully data-free
(dynamic activations, memoryless_minmax weights) and NVFP4 *weights* are data-free too.
32 samples ~ 140 s. The 512-sample GPTQ-style default projects to ~9.7 HOURS for no benefit.
"""
from __future__ import annotations
import argparse, json, time, torch
from transformers import AutoModelForCausalLM, AutoTokenizer
from llmcompressor import oneshot
from llmcompressor.modifiers.quantization import QuantizationModifier
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("src"); ap.add_argument("out")
ap.add_argument("--recipe", default="unsloth_qconfig.json")
ap.add_argument("--samples", type=int, default=32)
ap.add_argument("--seq", type=int, default=2048)
# registry key, NOT an HF repo id — 'HuggingFaceH4/ultrachat_200k' raises KeyError
ap.add_argument("--dataset", default="ultrachat-200k")
a = ap.parse_args()
q = json.load(open(a.recipe))
print("=== allocation (verbatim from the recipe file) ===", flush=True)
for g, v in q["config_groups"].items():
w = v["weights"]; act = v.get("input_activations") or {}
print(f" {g}: w={w['num_bits']}b/{w['type']}/{w['strategy']} gs={w.get('group_size')}"
f" | act dyn={act.get('dynamic')} | {len(v['targets'])} target patterns", flush=True)
print(f" ignore: {len(q['ignore'])} modules", flush=True)
t0 = time.time()
print(f"\n=== loading BF16 to CPU from {a.src} ===", flush=True)
model = AutoModelForCausalLM.from_pretrained(a.src, dtype=torch.bfloat16, device_map="cpu")
tok = AutoTokenizer.from_pretrained(a.src)
print(f"loaded in {time.time()-t0:.0f}s", flush=True)
recipe = QuantizationModifier(config_groups=q["config_groups"], ignore=q["ignore"])
print(f"\n=== oneshot: {a.samples} samples, seq {a.seq}, sequential pipeline ===", flush=True)
t1 = time.time()
oneshot(model=model, tokenizer=tok, dataset=a.dataset,
splits={"calibration": f"train_sft[:{a.samples}]"},
recipe=recipe, max_seq_length=a.seq, num_calibration_samples=a.samples,
pipeline="sequential", output_dir=a.out, save_compressed=True)
print(f"\noneshot done in {time.time()-t1:.0f}s (total {time.time()-t0:.0f}s)", flush=True)
print("QUANTIZATION COMPLETE", flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())