import json import os import shutil import subprocess import zipfile import coremltools as ct import numpy as np import torch import torch.nn as nn from transformers import AutoModelForCausalLM, AutoTokenizer MODEL_NAME = "HuggingFaceTB/SmolLM2-135M-Instruct" BUNDLE_NAME = "smollm2_135m_instruct_only_logits" SEQ_LEN = 32 MLPACKAGE_NAME = f"{BUNDLE_NAME}.mlpackage" COMPILED_DIR = "CompiledModel" COMPILED_NAME = f"{BUNDLE_NAME}.mlmodelc" ZIP_NAME = f"{COMPILED_NAME}.zip" TOKENIZER_DIR = "tokenizer" METADATA_NAME = "metadata.json" def safe_rm(path: str) -> None: if os.path.isdir(path): shutil.rmtree(path, ignore_errors=True) elif os.path.exists(path): os.remove(path) def write_runtime_metadata(tokenizer, output_path: str) -> None: metadata = { "runtime": "tinybrain-causallm-llama-v1", "model_id": MODEL_NAME, "bundle_name": BUNDLE_NAME, "architecture": "llama", "sequence_length": SEQ_LEN, "vocab_size": len(tokenizer), "input_names": ["input_ids", "attention_mask"], "output_name": "logits", "tokenizer_dir": "tokenizer", "requires_compiled_model": True, } with open(output_path, "w", encoding="utf-8") as f: json.dump(metadata, f, indent=2) for path in [ZIP_NAME, MLPACKAGE_NAME, COMPILED_DIR, TOKENIZER_DIR, METADATA_NAME]: safe_rm(path) tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) if tokenizer.pad_token is None: tokenizer.pad_token = tokenizer.eos_token model = AutoModelForCausalLM.from_pretrained(MODEL_NAME, use_safetensors=True) model.eval() class Wrapper(nn.Module): def __init__(self, base_model): super().__init__() self.base_model = base_model def forward(self, input_ids, attention_mask): input_ids = input_ids[:, :SEQ_LEN] attention_mask = attention_mask[:, :SEQ_LEN] outputs = self.base_model(input_ids=input_ids, attention_mask=attention_mask) return outputs.logits wrapped = Wrapper(model).eval() example = tokenizer( "hello coreml", return_tensors="pt", padding="max_length", truncation=True, max_length=SEQ_LEN, ) input_ids = example["input_ids"].to(dtype=torch.int32) attention_mask = example["attention_mask"].to(dtype=torch.int32) traced = torch.jit.trace(wrapped, (input_ids, attention_mask)) mlmodel = ct.convert( traced, inputs=[ ct.TensorType(name="input_ids", shape=(1, SEQ_LEN), dtype=np.int32), ct.TensorType(name="attention_mask", shape=(1, SEQ_LEN), dtype=np.int32), ], outputs=[ct.TensorType(name="logits")], convert_to="mlprogram", minimum_deployment_target=ct.target.iOS17, ) mlmodel.save(MLPACKAGE_NAME) subprocess.run(["xcrun", "coremlc", "compile", MLPACKAGE_NAME, COMPILED_DIR], check=True) compiled_path = os.path.join(COMPILED_DIR, COMPILED_NAME) tokenizer.save_pretrained(TOKENIZER_DIR) with open(os.path.join(TOKENIZER_DIR, "token_decoder.json"), "w", encoding="utf-8") as f: decoder = {str(token_id): token for token, token_id in tokenizer.get_vocab().items()} json.dump(decoder, f, ensure_ascii=False) write_runtime_metadata(tokenizer, METADATA_NAME) with zipfile.ZipFile(ZIP_NAME, "w", zipfile.ZIP_DEFLATED) as archive: for root, _, files in os.walk(compiled_path): for filename in files: full_path = os.path.join(root, filename) rel_path = os.path.relpath(full_path, compiled_path) archive.write(full_path, os.path.join(COMPILED_NAME, rel_path)) for root, _, files in os.walk(TOKENIZER_DIR): for filename in files: full_path = os.path.join(root, filename) rel_path = os.path.relpath(full_path, TOKENIZER_DIR) archive.write(full_path, os.path.join("tokenizer", rel_path)) archive.write(METADATA_NAME, METADATA_NAME) print(f"OK: {ZIP_NAME}")