Clemens Hemmerling commited on
Commit
7a1583f
·
1 Parent(s): 05026c6

Standardize TinyBrain bundle

Browse files
README.md CHANGED
@@ -1,20 +1,58 @@
1
  ---
2
  tags:
3
  - coreml
4
- - Apple
5
  - watchos
6
- - mlpackage
7
  library_name: coreml
8
  ---
9
 
10
- # TinyGPT2 for CoreML
11
 
12
- This is a distilled GPT-2 model converted to CoreML as an ML Program (`.mlpackage`),
13
- compatible with Apple Watch (watchOS 10+) and other Apple devices.
14
 
15
- - Base model: `sshleifer/tiny-gpt2`
16
- - Format: `.mlpackage`
17
- - Input: tokenized text (input_ids)
18
- - Output: logits for language modeling
19
- - Compatible with `coremltools` and Apple Watch
20
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
  tags:
3
  - coreml
4
+ - apple
5
  - watchos
6
+ - tinybrain
7
  library_name: coreml
8
  ---
9
 
10
+ # TinyGPT2 for TinyBrain
11
 
12
+ This repository packages `sshleifer/tiny-gpt2` for the TinyBrain runtime on Apple Watch.
 
13
 
14
+ ## TinyBrain bundle contract
 
 
 
 
15
 
16
+ The conversion script produces a single bundle archive:
17
+
18
+ - `tinygpt2_only_logits.mlmodelc.zip`
19
+
20
+ The zip layout is:
21
+
22
+ - `tinygpt2_only_logits.mlmodelc/`
23
+ - `tokenizer/`
24
+ - `metadata.json`
25
+
26
+ ## Runtime interface
27
+
28
+ - Task: causal language modeling
29
+ - Inputs:
30
+ - `input_ids` (`Int32`, shape `1 x 16`)
31
+ - `attention_mask` (`Int32`, shape `1 x 16`)
32
+ - Output:
33
+ - `logits`
34
+
35
+ ## Files included in `tokenizer/`
36
+
37
+ - `tokenizer.json` when provided by Transformers
38
+ - `vocab.json`
39
+ - `merges.txt`
40
+ - `special_tokens_map.json`
41
+ - `tokenizer_config.json`
42
+ - `token_decoder.json`
43
+
44
+ ## Build
45
+
46
+ Run:
47
+
48
+ ```bash
49
+ python3 convert_tinygpt2_to_coreml.py
50
+ ```
51
+
52
+ The script:
53
+
54
+ 1. Downloads the model and tokenizer from Hugging Face.
55
+ 2. Converts the model to Core ML.
56
+ 3. Compiles the package to `.mlmodelc`.
57
+ 4. Writes TinyBrain runtime metadata.
58
+ 5. Packages the compiled model, tokenizer, and metadata into one zip.
convert_tinygpt2_to_coreml.py ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import json
2
+ import os
3
+ import shutil
4
+ import subprocess
5
+ import zipfile
6
+
7
+ import coremltools as ct
8
+ import numpy as np
9
+ import torch
10
+ import torch.nn as nn
11
+ from transformers import AutoModelForCausalLM, AutoTokenizer
12
+
13
+ MODEL_NAME = "sshleifer/tiny-gpt2"
14
+ BUNDLE_NAME = "tinygpt2_only_logits"
15
+ SEQ_LEN = 16
16
+
17
+ MLPACKAGE_NAME = f"{BUNDLE_NAME}.mlpackage"
18
+ COMPILED_DIR = "CompiledModel"
19
+ COMPILED_NAME = f"{BUNDLE_NAME}.mlmodelc"
20
+ ZIP_NAME = f"{COMPILED_NAME}.zip"
21
+ TOKENIZER_DIR = "tokenizer"
22
+ METADATA_NAME = "metadata.json"
23
+
24
+
25
+ def safe_rm(path: str) -> None:
26
+ if os.path.isdir(path):
27
+ shutil.rmtree(path, ignore_errors=True)
28
+ elif os.path.exists(path):
29
+ os.remove(path)
30
+
31
+
32
+ def write_runtime_metadata(tokenizer, output_path: str) -> None:
33
+ metadata = {
34
+ "runtime": "tinybrain-causallm-v1",
35
+ "model_id": MODEL_NAME,
36
+ "bundle_name": BUNDLE_NAME,
37
+ "architecture": "gpt2",
38
+ "sequence_length": SEQ_LEN,
39
+ "vocab_size": len(tokenizer),
40
+ "input_names": ["input_ids", "attention_mask"],
41
+ "output_name": "logits",
42
+ "tokenizer_dir": "tokenizer",
43
+ "requires_compiled_model": True,
44
+ }
45
+ with open(output_path, "w", encoding="utf-8") as f:
46
+ json.dump(metadata, f, indent=2)
47
+
48
+
49
+ for path in [ZIP_NAME, MLPACKAGE_NAME, COMPILED_DIR, TOKENIZER_DIR, METADATA_NAME]:
50
+ safe_rm(path)
51
+
52
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
53
+ if tokenizer.pad_token is None:
54
+ tokenizer.pad_token = tokenizer.eos_token
55
+
56
+ model = AutoModelForCausalLM.from_pretrained(MODEL_NAME, use_safetensors=True)
57
+ model.eval()
58
+
59
+
60
+ class Wrapper(nn.Module):
61
+ def __init__(self, base_model):
62
+ super().__init__()
63
+ self.base_model = base_model
64
+
65
+ def forward(self, input_ids, attention_mask):
66
+ input_ids = input_ids[:, :SEQ_LEN]
67
+ attention_mask = attention_mask[:, :SEQ_LEN]
68
+ outputs = self.base_model(input_ids=input_ids, attention_mask=attention_mask)
69
+ return outputs.logits
70
+
71
+
72
+ wrapped = Wrapper(model).eval()
73
+ example = tokenizer(
74
+ "hello coreml",
75
+ return_tensors="pt",
76
+ padding="max_length",
77
+ truncation=True,
78
+ max_length=SEQ_LEN,
79
+ )
80
+ input_ids = example["input_ids"].to(dtype=torch.int32)
81
+ attention_mask = example["attention_mask"].to(dtype=torch.int32)
82
+
83
+ traced = torch.jit.trace(wrapped, (input_ids, attention_mask))
84
+ mlmodel = ct.convert(
85
+ traced,
86
+ inputs=[
87
+ ct.TensorType(name="input_ids", shape=(1, SEQ_LEN), dtype=np.int32),
88
+ ct.TensorType(name="attention_mask", shape=(1, SEQ_LEN), dtype=np.int32),
89
+ ],
90
+ outputs=[ct.TensorType(name="logits")],
91
+ convert_to="mlprogram",
92
+ minimum_deployment_target=ct.target.iOS17,
93
+ )
94
+ mlmodel.save(MLPACKAGE_NAME)
95
+
96
+ subprocess.run(["xcrun", "coremlc", "compile", MLPACKAGE_NAME, COMPILED_DIR], check=True)
97
+ compiled_path = os.path.join(COMPILED_DIR, COMPILED_NAME)
98
+
99
+ tokenizer.save_pretrained(TOKENIZER_DIR)
100
+ with open(os.path.join(TOKENIZER_DIR, "token_decoder.json"), "w", encoding="utf-8") as f:
101
+ decoder = {str(token_id): token for token, token_id in tokenizer.get_vocab().items()}
102
+ json.dump(decoder, f, ensure_ascii=False)
103
+
104
+ write_runtime_metadata(tokenizer, METADATA_NAME)
105
+
106
+ with zipfile.ZipFile(ZIP_NAME, "w", zipfile.ZIP_DEFLATED) as archive:
107
+ for root, _, files in os.walk(compiled_path):
108
+ for filename in files:
109
+ full_path = os.path.join(root, filename)
110
+ rel_path = os.path.relpath(full_path, compiled_path)
111
+ archive.write(full_path, os.path.join(COMPILED_NAME, rel_path))
112
+ for root, _, files in os.walk(TOKENIZER_DIR):
113
+ for filename in files:
114
+ full_path = os.path.join(root, filename)
115
+ rel_path = os.path.relpath(full_path, TOKENIZER_DIR)
116
+ archive.write(full_path, os.path.join("tokenizer", rel_path))
117
+ archive.write(METADATA_NAME, METADATA_NAME)
118
+
119
+ print(f"OK: {ZIP_NAME}")
metadata.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "runtime": "tinybrain-causallm-v1",
3
+ "model_id": "sshleifer/tiny-gpt2",
4
+ "bundle_name": "tinygpt2_only_logits",
5
+ "architecture": "gpt2",
6
+ "sequence_length": 16,
7
+ "vocab_size": 50257,
8
+ "input_names": [
9
+ "input_ids",
10
+ "attention_mask"
11
+ ],
12
+ "output_name": "logits",
13
+ "tokenizer_dir": "tokenizer",
14
+ "requires_compiled_model": true
15
+ }
tinygpt2_only_logits.mlmodelc.zip ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6552b89db61be95a22fb94961c5b404ade2011c1a5abeaabce383c741dbb61ef
3
+ size 1773389
tokenizer/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer/special_tokens_map.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|endoftext|>",
4
+ "lstrip": false,
5
+ "normalized": true,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|endoftext|>",
11
+ "lstrip": false,
12
+ "normalized": true,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": "<|endoftext|>",
17
+ "unk_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": true,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ }
24
+ }
tokenizer/token_decoder.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer/tokenizer_config.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "50256": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": true,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ }
13
+ },
14
+ "bos_token": "<|endoftext|>",
15
+ "clean_up_tokenization_spaces": false,
16
+ "eos_token": "<|endoftext|>",
17
+ "errors": "replace",
18
+ "extra_special_tokens": {},
19
+ "model_max_length": 1024,
20
+ "pad_token": "<|endoftext|>",
21
+ "tokenizer_class": "GPT2Tokenizer",
22
+ "unk_token": "<|endoftext|>"
23
+ }
tokenizer/vocab.json ADDED
The diff for this file is too large to render. See raw diff