File size: 3,883 Bytes
5bec923 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 | import json
import os
import shutil
import subprocess
import zipfile
import coremltools as ct
import numpy as np
import torch
import torch.nn as nn
from transformers import AutoModelForCausalLM, AutoTokenizer
MODEL_NAME = "HuggingFaceTB/SmolLM2-135M-Instruct"
BUNDLE_NAME = "smollm2_135m_instruct_only_logits"
SEQ_LEN = 32
MLPACKAGE_NAME = f"{BUNDLE_NAME}.mlpackage"
COMPILED_DIR = "CompiledModel"
COMPILED_NAME = f"{BUNDLE_NAME}.mlmodelc"
ZIP_NAME = f"{COMPILED_NAME}.zip"
TOKENIZER_DIR = "tokenizer"
METADATA_NAME = "metadata.json"
def safe_rm(path: str) -> None:
if os.path.isdir(path):
shutil.rmtree(path, ignore_errors=True)
elif os.path.exists(path):
os.remove(path)
def write_runtime_metadata(tokenizer, output_path: str) -> None:
metadata = {
"runtime": "tinybrain-causallm-llama-v1",
"model_id": MODEL_NAME,
"bundle_name": BUNDLE_NAME,
"architecture": "llama",
"sequence_length": SEQ_LEN,
"vocab_size": len(tokenizer),
"input_names": ["input_ids", "attention_mask"],
"output_name": "logits",
"tokenizer_dir": "tokenizer",
"requires_compiled_model": True,
}
with open(output_path, "w", encoding="utf-8") as f:
json.dump(metadata, f, indent=2)
for path in [ZIP_NAME, MLPACKAGE_NAME, COMPILED_DIR, TOKENIZER_DIR, METADATA_NAME]:
safe_rm(path)
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token
model = AutoModelForCausalLM.from_pretrained(MODEL_NAME, use_safetensors=True)
model.eval()
class Wrapper(nn.Module):
def __init__(self, base_model):
super().__init__()
self.base_model = base_model
def forward(self, input_ids, attention_mask):
input_ids = input_ids[:, :SEQ_LEN]
attention_mask = attention_mask[:, :SEQ_LEN]
outputs = self.base_model(input_ids=input_ids, attention_mask=attention_mask)
return outputs.logits
wrapped = Wrapper(model).eval()
example = tokenizer(
"hello coreml",
return_tensors="pt",
padding="max_length",
truncation=True,
max_length=SEQ_LEN,
)
input_ids = example["input_ids"].to(dtype=torch.int32)
attention_mask = example["attention_mask"].to(dtype=torch.int32)
traced = torch.jit.trace(wrapped, (input_ids, attention_mask))
mlmodel = ct.convert(
traced,
inputs=[
ct.TensorType(name="input_ids", shape=(1, SEQ_LEN), dtype=np.int32),
ct.TensorType(name="attention_mask", shape=(1, SEQ_LEN), dtype=np.int32),
],
outputs=[ct.TensorType(name="logits")],
convert_to="mlprogram",
minimum_deployment_target=ct.target.iOS17,
)
mlmodel.save(MLPACKAGE_NAME)
subprocess.run(["xcrun", "coremlc", "compile", MLPACKAGE_NAME, COMPILED_DIR], check=True)
compiled_path = os.path.join(COMPILED_DIR, COMPILED_NAME)
tokenizer.save_pretrained(TOKENIZER_DIR)
with open(os.path.join(TOKENIZER_DIR, "token_decoder.json"), "w", encoding="utf-8") as f:
decoder = {str(token_id): token for token, token_id in tokenizer.get_vocab().items()}
json.dump(decoder, f, ensure_ascii=False)
write_runtime_metadata(tokenizer, METADATA_NAME)
with zipfile.ZipFile(ZIP_NAME, "w", zipfile.ZIP_DEFLATED) as archive:
for root, _, files in os.walk(compiled_path):
for filename in files:
full_path = os.path.join(root, filename)
rel_path = os.path.relpath(full_path, compiled_path)
archive.write(full_path, os.path.join(COMPILED_NAME, rel_path))
for root, _, files in os.walk(TOKENIZER_DIR):
for filename in files:
full_path = os.path.join(root, filename)
rel_path = os.path.relpath(full_path, TOKENIZER_DIR)
archive.write(full_path, os.path.join("tokenizer", rel_path))
archive.write(METADATA_NAME, METADATA_NAME)
print(f"OK: {ZIP_NAME}")
|