smollm2-135m-instruct-coreml / convert_smollm2_135m_instruct_to_coreml.py
clemenshemmerling's picture
Add files using upload-large-folder tool
5bec923 verified
Raw History Blame Contribute Delete
3.88 kB
import json
import os
import shutil
import subprocess
import zipfile
import coremltools as ct
import numpy as np
import torch
import torch.nn as nn
from transformers import AutoModelForCausalLM, AutoTokenizer
MODEL_NAME = "HuggingFaceTB/SmolLM2-135M-Instruct"
BUNDLE_NAME = "smollm2_135m_instruct_only_logits"
SEQ_LEN = 32
MLPACKAGE_NAME = f"{BUNDLE_NAME}.mlpackage"
COMPILED_DIR = "CompiledModel"
COMPILED_NAME = f"{BUNDLE_NAME}.mlmodelc"
ZIP_NAME = f"{COMPILED_NAME}.zip"
TOKENIZER_DIR = "tokenizer"
METADATA_NAME = "metadata.json"
def safe_rm(path: str) -> None:
if os.path.isdir(path):
shutil.rmtree(path, ignore_errors=True)
elif os.path.exists(path):
os.remove(path)
def write_runtime_metadata(tokenizer, output_path: str) -> None:
metadata = {
"runtime": "tinybrain-causallm-llama-v1",
"model_id": MODEL_NAME,
"bundle_name": BUNDLE_NAME,
"architecture": "llama",
"sequence_length": SEQ_LEN,
"vocab_size": len(tokenizer),
"input_names": ["input_ids", "attention_mask"],
"output_name": "logits",
"tokenizer_dir": "tokenizer",
"requires_compiled_model": True,
}
with open(output_path, "w", encoding="utf-8") as f:
json.dump(metadata, f, indent=2)
for path in [ZIP_NAME, MLPACKAGE_NAME, COMPILED_DIR, TOKENIZER_DIR, METADATA_NAME]:
safe_rm(path)
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token
model = AutoModelForCausalLM.from_pretrained(MODEL_NAME, use_safetensors=True)
model.eval()
class Wrapper(nn.Module):
def __init__(self, base_model):
super().__init__()
self.base_model = base_model
def forward(self, input_ids, attention_mask):
input_ids = input_ids[:, :SEQ_LEN]
attention_mask = attention_mask[:, :SEQ_LEN]
outputs = self.base_model(input_ids=input_ids, attention_mask=attention_mask)
return outputs.logits
wrapped = Wrapper(model).eval()
example = tokenizer(
"hello coreml",
return_tensors="pt",
padding="max_length",
truncation=True,
max_length=SEQ_LEN,
)
input_ids = example["input_ids"].to(dtype=torch.int32)
attention_mask = example["attention_mask"].to(dtype=torch.int32)
traced = torch.jit.trace(wrapped, (input_ids, attention_mask))
mlmodel = ct.convert(
traced,
inputs=[
ct.TensorType(name="input_ids", shape=(1, SEQ_LEN), dtype=np.int32),
ct.TensorType(name="attention_mask", shape=(1, SEQ_LEN), dtype=np.int32),
],
outputs=[ct.TensorType(name="logits")],
convert_to="mlprogram",
minimum_deployment_target=ct.target.iOS17,
)
mlmodel.save(MLPACKAGE_NAME)
subprocess.run(["xcrun", "coremlc", "compile", MLPACKAGE_NAME, COMPILED_DIR], check=True)
compiled_path = os.path.join(COMPILED_DIR, COMPILED_NAME)
tokenizer.save_pretrained(TOKENIZER_DIR)
with open(os.path.join(TOKENIZER_DIR, "token_decoder.json"), "w", encoding="utf-8") as f:
decoder = {str(token_id): token for token, token_id in tokenizer.get_vocab().items()}
json.dump(decoder, f, ensure_ascii=False)
write_runtime_metadata(tokenizer, METADATA_NAME)
with zipfile.ZipFile(ZIP_NAME, "w", zipfile.ZIP_DEFLATED) as archive:
for root, _, files in os.walk(compiled_path):
for filename in files:
full_path = os.path.join(root, filename)
rel_path = os.path.relpath(full_path, compiled_path)
archive.write(full_path, os.path.join(COMPILED_NAME, rel_path))
for root, _, files in os.walk(TOKENIZER_DIR):
for filename in files:
full_path = os.path.join(root, filename)
rel_path = os.path.relpath(full_path, TOKENIZER_DIR)
archive.write(full_path, os.path.join("tokenizer", rel_path))
archive.write(METADATA_NAME, METADATA_NAME)
print(f"OK: {ZIP_NAME}")