Clemens Hemmerling commited on
Commit ·
7a1583f
1
Parent(s): 05026c6
Standardize TinyBrain bundle
Browse files- README.md +48 -10
- convert_tinygpt2_to_coreml.py +119 -0
- metadata.json +15 -0
- tinygpt2_only_logits.mlmodelc.zip +3 -0
- tokenizer/merges.txt +0 -0
- tokenizer/special_tokens_map.json +24 -0
- tokenizer/token_decoder.json +0 -0
- tokenizer/tokenizer.json +0 -0
- tokenizer/tokenizer_config.json +23 -0
- tokenizer/vocab.json +0 -0
README.md
CHANGED
|
@@ -1,20 +1,58 @@
|
|
| 1 |
---
|
| 2 |
tags:
|
| 3 |
- coreml
|
| 4 |
-
-
|
| 5 |
- watchos
|
| 6 |
-
-
|
| 7 |
library_name: coreml
|
| 8 |
---
|
| 9 |
|
| 10 |
-
# TinyGPT2 for
|
| 11 |
|
| 12 |
-
This
|
| 13 |
-
compatible with Apple Watch (watchOS 10+) and other Apple devices.
|
| 14 |
|
| 15 |
-
|
| 16 |
-
- Format: `.mlpackage`
|
| 17 |
-
- Input: tokenized text (input_ids)
|
| 18 |
-
- Output: logits for language modeling
|
| 19 |
-
- Compatible with `coremltools` and Apple Watch
|
| 20 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
tags:
|
| 3 |
- coreml
|
| 4 |
+
- apple
|
| 5 |
- watchos
|
| 6 |
+
- tinybrain
|
| 7 |
library_name: coreml
|
| 8 |
---
|
| 9 |
|
| 10 |
+
# TinyGPT2 for TinyBrain
|
| 11 |
|
| 12 |
+
This repository packages `sshleifer/tiny-gpt2` for the TinyBrain runtime on Apple Watch.
|
|
|
|
| 13 |
|
| 14 |
+
## TinyBrain bundle contract
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
|
| 16 |
+
The conversion script produces a single bundle archive:
|
| 17 |
+
|
| 18 |
+
- `tinygpt2_only_logits.mlmodelc.zip`
|
| 19 |
+
|
| 20 |
+
The zip layout is:
|
| 21 |
+
|
| 22 |
+
- `tinygpt2_only_logits.mlmodelc/`
|
| 23 |
+
- `tokenizer/`
|
| 24 |
+
- `metadata.json`
|
| 25 |
+
|
| 26 |
+
## Runtime interface
|
| 27 |
+
|
| 28 |
+
- Task: causal language modeling
|
| 29 |
+
- Inputs:
|
| 30 |
+
- `input_ids` (`Int32`, shape `1 x 16`)
|
| 31 |
+
- `attention_mask` (`Int32`, shape `1 x 16`)
|
| 32 |
+
- Output:
|
| 33 |
+
- `logits`
|
| 34 |
+
|
| 35 |
+
## Files included in `tokenizer/`
|
| 36 |
+
|
| 37 |
+
- `tokenizer.json` when provided by Transformers
|
| 38 |
+
- `vocab.json`
|
| 39 |
+
- `merges.txt`
|
| 40 |
+
- `special_tokens_map.json`
|
| 41 |
+
- `tokenizer_config.json`
|
| 42 |
+
- `token_decoder.json`
|
| 43 |
+
|
| 44 |
+
## Build
|
| 45 |
+
|
| 46 |
+
Run:
|
| 47 |
+
|
| 48 |
+
```bash
|
| 49 |
+
python3 convert_tinygpt2_to_coreml.py
|
| 50 |
+
```
|
| 51 |
+
|
| 52 |
+
The script:
|
| 53 |
+
|
| 54 |
+
1. Downloads the model and tokenizer from Hugging Face.
|
| 55 |
+
2. Converts the model to Core ML.
|
| 56 |
+
3. Compiles the package to `.mlmodelc`.
|
| 57 |
+
4. Writes TinyBrain runtime metadata.
|
| 58 |
+
5. Packages the compiled model, tokenizer, and metadata into one zip.
|
convert_tinygpt2_to_coreml.py
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import json
|
| 2 |
+
import os
|
| 3 |
+
import shutil
|
| 4 |
+
import subprocess
|
| 5 |
+
import zipfile
|
| 6 |
+
|
| 7 |
+
import coremltools as ct
|
| 8 |
+
import numpy as np
|
| 9 |
+
import torch
|
| 10 |
+
import torch.nn as nn
|
| 11 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 12 |
+
|
| 13 |
+
MODEL_NAME = "sshleifer/tiny-gpt2"
|
| 14 |
+
BUNDLE_NAME = "tinygpt2_only_logits"
|
| 15 |
+
SEQ_LEN = 16
|
| 16 |
+
|
| 17 |
+
MLPACKAGE_NAME = f"{BUNDLE_NAME}.mlpackage"
|
| 18 |
+
COMPILED_DIR = "CompiledModel"
|
| 19 |
+
COMPILED_NAME = f"{BUNDLE_NAME}.mlmodelc"
|
| 20 |
+
ZIP_NAME = f"{COMPILED_NAME}.zip"
|
| 21 |
+
TOKENIZER_DIR = "tokenizer"
|
| 22 |
+
METADATA_NAME = "metadata.json"
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def safe_rm(path: str) -> None:
|
| 26 |
+
if os.path.isdir(path):
|
| 27 |
+
shutil.rmtree(path, ignore_errors=True)
|
| 28 |
+
elif os.path.exists(path):
|
| 29 |
+
os.remove(path)
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def write_runtime_metadata(tokenizer, output_path: str) -> None:
|
| 33 |
+
metadata = {
|
| 34 |
+
"runtime": "tinybrain-causallm-v1",
|
| 35 |
+
"model_id": MODEL_NAME,
|
| 36 |
+
"bundle_name": BUNDLE_NAME,
|
| 37 |
+
"architecture": "gpt2",
|
| 38 |
+
"sequence_length": SEQ_LEN,
|
| 39 |
+
"vocab_size": len(tokenizer),
|
| 40 |
+
"input_names": ["input_ids", "attention_mask"],
|
| 41 |
+
"output_name": "logits",
|
| 42 |
+
"tokenizer_dir": "tokenizer",
|
| 43 |
+
"requires_compiled_model": True,
|
| 44 |
+
}
|
| 45 |
+
with open(output_path, "w", encoding="utf-8") as f:
|
| 46 |
+
json.dump(metadata, f, indent=2)
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
for path in [ZIP_NAME, MLPACKAGE_NAME, COMPILED_DIR, TOKENIZER_DIR, METADATA_NAME]:
|
| 50 |
+
safe_rm(path)
|
| 51 |
+
|
| 52 |
+
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
|
| 53 |
+
if tokenizer.pad_token is None:
|
| 54 |
+
tokenizer.pad_token = tokenizer.eos_token
|
| 55 |
+
|
| 56 |
+
model = AutoModelForCausalLM.from_pretrained(MODEL_NAME, use_safetensors=True)
|
| 57 |
+
model.eval()
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
class Wrapper(nn.Module):
|
| 61 |
+
def __init__(self, base_model):
|
| 62 |
+
super().__init__()
|
| 63 |
+
self.base_model = base_model
|
| 64 |
+
|
| 65 |
+
def forward(self, input_ids, attention_mask):
|
| 66 |
+
input_ids = input_ids[:, :SEQ_LEN]
|
| 67 |
+
attention_mask = attention_mask[:, :SEQ_LEN]
|
| 68 |
+
outputs = self.base_model(input_ids=input_ids, attention_mask=attention_mask)
|
| 69 |
+
return outputs.logits
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
wrapped = Wrapper(model).eval()
|
| 73 |
+
example = tokenizer(
|
| 74 |
+
"hello coreml",
|
| 75 |
+
return_tensors="pt",
|
| 76 |
+
padding="max_length",
|
| 77 |
+
truncation=True,
|
| 78 |
+
max_length=SEQ_LEN,
|
| 79 |
+
)
|
| 80 |
+
input_ids = example["input_ids"].to(dtype=torch.int32)
|
| 81 |
+
attention_mask = example["attention_mask"].to(dtype=torch.int32)
|
| 82 |
+
|
| 83 |
+
traced = torch.jit.trace(wrapped, (input_ids, attention_mask))
|
| 84 |
+
mlmodel = ct.convert(
|
| 85 |
+
traced,
|
| 86 |
+
inputs=[
|
| 87 |
+
ct.TensorType(name="input_ids", shape=(1, SEQ_LEN), dtype=np.int32),
|
| 88 |
+
ct.TensorType(name="attention_mask", shape=(1, SEQ_LEN), dtype=np.int32),
|
| 89 |
+
],
|
| 90 |
+
outputs=[ct.TensorType(name="logits")],
|
| 91 |
+
convert_to="mlprogram",
|
| 92 |
+
minimum_deployment_target=ct.target.iOS17,
|
| 93 |
+
)
|
| 94 |
+
mlmodel.save(MLPACKAGE_NAME)
|
| 95 |
+
|
| 96 |
+
subprocess.run(["xcrun", "coremlc", "compile", MLPACKAGE_NAME, COMPILED_DIR], check=True)
|
| 97 |
+
compiled_path = os.path.join(COMPILED_DIR, COMPILED_NAME)
|
| 98 |
+
|
| 99 |
+
tokenizer.save_pretrained(TOKENIZER_DIR)
|
| 100 |
+
with open(os.path.join(TOKENIZER_DIR, "token_decoder.json"), "w", encoding="utf-8") as f:
|
| 101 |
+
decoder = {str(token_id): token for token, token_id in tokenizer.get_vocab().items()}
|
| 102 |
+
json.dump(decoder, f, ensure_ascii=False)
|
| 103 |
+
|
| 104 |
+
write_runtime_metadata(tokenizer, METADATA_NAME)
|
| 105 |
+
|
| 106 |
+
with zipfile.ZipFile(ZIP_NAME, "w", zipfile.ZIP_DEFLATED) as archive:
|
| 107 |
+
for root, _, files in os.walk(compiled_path):
|
| 108 |
+
for filename in files:
|
| 109 |
+
full_path = os.path.join(root, filename)
|
| 110 |
+
rel_path = os.path.relpath(full_path, compiled_path)
|
| 111 |
+
archive.write(full_path, os.path.join(COMPILED_NAME, rel_path))
|
| 112 |
+
for root, _, files in os.walk(TOKENIZER_DIR):
|
| 113 |
+
for filename in files:
|
| 114 |
+
full_path = os.path.join(root, filename)
|
| 115 |
+
rel_path = os.path.relpath(full_path, TOKENIZER_DIR)
|
| 116 |
+
archive.write(full_path, os.path.join("tokenizer", rel_path))
|
| 117 |
+
archive.write(METADATA_NAME, METADATA_NAME)
|
| 118 |
+
|
| 119 |
+
print(f"OK: {ZIP_NAME}")
|
metadata.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"runtime": "tinybrain-causallm-v1",
|
| 3 |
+
"model_id": "sshleifer/tiny-gpt2",
|
| 4 |
+
"bundle_name": "tinygpt2_only_logits",
|
| 5 |
+
"architecture": "gpt2",
|
| 6 |
+
"sequence_length": 16,
|
| 7 |
+
"vocab_size": 50257,
|
| 8 |
+
"input_names": [
|
| 9 |
+
"input_ids",
|
| 10 |
+
"attention_mask"
|
| 11 |
+
],
|
| 12 |
+
"output_name": "logits",
|
| 13 |
+
"tokenizer_dir": "tokenizer",
|
| 14 |
+
"requires_compiled_model": true
|
| 15 |
+
}
|
tinygpt2_only_logits.mlmodelc.zip
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6552b89db61be95a22fb94961c5b404ade2011c1a5abeaabce383c741dbb61ef
|
| 3 |
+
size 1773389
|
tokenizer/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer/special_tokens_map.json
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "<|endoftext|>",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": true,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"eos_token": {
|
| 10 |
+
"content": "<|endoftext|>",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": true,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": "<|endoftext|>",
|
| 17 |
+
"unk_token": {
|
| 18 |
+
"content": "<|endoftext|>",
|
| 19 |
+
"lstrip": false,
|
| 20 |
+
"normalized": true,
|
| 21 |
+
"rstrip": false,
|
| 22 |
+
"single_word": false
|
| 23 |
+
}
|
| 24 |
+
}
|
tokenizer/token_decoder.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer/tokenizer_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_prefix_space": false,
|
| 4 |
+
"added_tokens_decoder": {
|
| 5 |
+
"50256": {
|
| 6 |
+
"content": "<|endoftext|>",
|
| 7 |
+
"lstrip": false,
|
| 8 |
+
"normalized": true,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false,
|
| 11 |
+
"special": true
|
| 12 |
+
}
|
| 13 |
+
},
|
| 14 |
+
"bos_token": "<|endoftext|>",
|
| 15 |
+
"clean_up_tokenization_spaces": false,
|
| 16 |
+
"eos_token": "<|endoftext|>",
|
| 17 |
+
"errors": "replace",
|
| 18 |
+
"extra_special_tokens": {},
|
| 19 |
+
"model_max_length": 1024,
|
| 20 |
+
"pad_token": "<|endoftext|>",
|
| 21 |
+
"tokenizer_class": "GPT2Tokenizer",
|
| 22 |
+
"unk_token": "<|endoftext|>"
|
| 23 |
+
}
|
tokenizer/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|