"""npu-forge-tune: THE pipeline — LoRA fine-tune -> merge -> in-job voice proof -> GGUF -> Q4NX, one container, one command. Output lands on the npu-forge-out volume ready for bin/assemble.js on the local box. modal run tune_npu.py (defaults: Llama-3.2-1B + grandma.jsonl) """ import modal app = modal.App("npu-forge-tune") image = ( modal.Image.debian_slim(python_version="3.11") .apt_install("git") .pip_install( "torch==2.4.1", "transformers==4.46.3", "peft==0.12.0", "trl==0.9.6", "datasets==2.21.0", "accelerate==1.1.1", "numpy", "gguf", "amd-quark", "huggingface_hub", "safetensors", "sentencepiece", "einops", "tqdm", "protobuf", ) .apt_install("cmake", "build-essential", "libcurl4-openssl-dev") .run_commands( "git clone --depth 1 https://github.com/FastFlowLM/FLM_Q4NX_Converter /converter", "git clone --depth 1 https://github.com/ggml-org/llama.cpp /llamacpp", # build just the quantize tool — needed to make Q4_K_M (the only GGUF # format proven to convert to a COHERENT Q4NX model on the NPU; the # converter's llama path rejects f16 and q8_0 double-quantizes to garbage) "cd /llamacpp && cmake -B build -DLLAMA_CURL=OFF -DGGML_NATIVE=OFF && cmake --build build --target llama-quantize -j 8", ) .add_local_dir("C:/Users/Forgemind/Desktop/npu-forge/scratch/tune-data", remote_path="/data") ) vol = modal.Volume.from_name("npu-forge-out", create_if_missing=True) @app.function(image=image, gpu="T4", cpu=8, memory=32768, timeout=5400, volumes={"/out": vol}) def tune(base: str, data_file: str, out_name: str, epochs: int = 3, probe: str = "Good morning grandma, how is the garden today?"): import json, os, subprocess, time import torch from datasets import load_dataset from transformers import AutoTokenizer, AutoModelForCausalLM from peft import LoraConfig from trl import SFTTrainer, SFTConfig timings = {} t = time.time() tok = AutoTokenizer.from_pretrained(base) if tok.pad_token is None: tok.pad_token = tok.eos_token model = AutoModelForCausalLM.from_pretrained(base, torch_dtype=torch.float16, device_map="cuda") ds = load_dataset("json", data_files=f"/data/{data_file}", split="train") ds = ds.map(lambda r: {"text": tok.apply_chat_template(r["messages"], tokenize=False)}) print(f"[data] {len(ds)} examples") trainer = SFTTrainer( model=model, train_dataset=ds, peft_config=LoraConfig(r=16, lora_alpha=32, lora_dropout=0.05, target_modules=["q_proj", "k_proj", "v_proj", "o_proj"]), args=SFTConfig( output_dir="/tmp/sft", num_train_epochs=epochs, per_device_train_batch_size=1, gradient_accumulation_steps=8, learning_rate=2e-4, logging_steps=20, max_seq_length=1024, dataset_text_field="text", report_to=[], save_strategy="no", fp16=True, gradient_checkpointing=True, ), ) trainer.train() timings["train_s"] = round(time.time() - t); t = time.time() merged = trainer.model.merge_and_unload() merged.save_pretrained("/tmp/merged", safe_serialization=True) tok.save_pretrained("/tmp/merged") timings["merge_s"] = round(time.time() - t); t = time.time() # in-job voice proof: does the merged model actually speak the tune? msgs = [{"role": "user", "content": probe}] inp = tok.apply_chat_template(msgs, add_generation_prompt=True, return_tensors="pt").to("cuda") with torch.no_grad(): gen = merged.generate(inp, max_new_tokens=160, temperature=0.8, do_sample=True, pad_token_id=tok.eos_token_id) sample = tok.decode(gen[0][inp.shape[1]:], skip_special_tokens=True) print("[voice proof] " + sample[:600]) timings["sample_s"] = round(time.time() - t); t = time.time() # HF -> f16 GGUF -> Q4_K_M GGUF -> Q4NX. The middle step matters: the # Q4NX converter's llama path needs pre-quantized blocks (rejects f16), # and Q4_K_M is the only format proven to yield a COHERENT NPU model # (q8_0 double-quantized to repetition garbage). Snags #9 & #10. r = subprocess.run(["python", "/llamacpp/convert_hf_to_gguf.py", "/tmp/merged", "--outfile", "/tmp/model-f16.gguf", "--outtype", "f16"], capture_output=True, text=True) if r.returncode != 0: print(r.stdout[-2000:]); print("STDERR:", r.stderr[-3000:]); raise RuntimeError("convert_hf_to_gguf failed") q = subprocess.run(["/llamacpp/build/bin/llama-quantize", "/tmp/model-f16.gguf", "/tmp/model-q4km.gguf", "Q4_K_M"], capture_output=True, text=True) if q.returncode != 0 or not os.path.exists("/tmp/model-q4km.gguf"): print(q.stdout[-1500:]); print("STDERR:", q.stderr[-2500:]); raise RuntimeError("llama-quantize failed") timings["gguf_s"] = round(time.time() - t); t = time.time() # GGUF -> Q4NX (proven stage: module API, cwd=/converter) import sys sys.path.insert(0, "/converter") os.chdir("/converter") from q4nx import create_converter outdir = f"/out/{out_name}" os.makedirs(outdir, exist_ok=True) create_converter("/tmp/model-q4km.gguf", "").convert(q4nx_path=outdir, weights_type="language") timings["q4nx_s"] = round(time.time() - t) files = {f: os.path.getsize(os.path.join(outdir, f)) for f in os.listdir(outdir)} with open(os.path.join(outdir, "tune-report.json"), "w") as fh: json.dump({"base": base, "data": data_file, "epochs": epochs, "sample": sample, "timings": timings}, fh, indent=2) vol.commit() return {"timings": timings, "outputs": files, "voice_sample": sample[:400]} @app.local_entrypoint() def main(base: str = "unsloth/Llama-3.2-1B-Instruct", data_file: str = "grandma.jsonl", out_name: str = "forge-grandma-1b", epochs: int = 3): import json print(json.dumps(tune.remote(base, data_file, out_name, epochs), indent=2))