Upload train_tokenizer.py with huggingface_hub
Browse files- train_tokenizer.py +73 -0
train_tokenizer.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""train_tokenizer.py - GoLLeM-v5 (EN) tokenizer training recipe.
|
| 3 |
+
|
| 4 |
+
Reproduces the canonical `tokenizer.json` shipped in this repo:
|
| 5 |
+
- byte-level BPE (GPT-2 lineage), pre-tokenizer + decoder = ByteLevel
|
| 6 |
+
- model vocab 12285 (256 byte alphabet + 12029 learned merges)
|
| 7 |
+
- 3 special tokens appended: <|endoftext|> <|im_start|> <|im_end|>
|
| 8 |
+
(=> 12288 effective ids; models pad the embedding to 12288)
|
| 9 |
+
- trained on the minimal-en-corpus (FineWeb-Edu EN broad mix; `en.parquet`,
|
| 10 |
+
the same source uploaded as SlayerLab/minimal-en-corpus-5b).
|
| 11 |
+
|
| 12 |
+
The shipped `tokenizer.json` remains the canonical / authoritative artifact:
|
| 13 |
+
a fresh run reproduces a functionally-equivalent tokenizer, but exact merge
|
| 14 |
+
order depends on the corpus snapshot/order, so byte-identity is not guaranteed.
|
| 15 |
+
Use this script to audit the method; use `tokenizer.json` for exact parity.
|
| 16 |
+
|
| 17 |
+
Usage:
|
| 18 |
+
python train_tokenizer.py --corpus en.parquet --out tokenizer.json
|
| 19 |
+
python train_tokenizer.py --limit 50000 --out tok_smoke.json # quick recipe check
|
| 20 |
+
"""
|
| 21 |
+
import argparse
|
| 22 |
+
import pyarrow.parquet as pq
|
| 23 |
+
from tokenizers import Tokenizer, models, trainers, pre_tokenizers, decoders
|
| 24 |
+
|
| 25 |
+
SPECIAL = ["<|endoftext|>", "<|im_start|>", "<|im_end|>"]
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def iter_text(parquet_path, col="text", limit=None):
|
| 29 |
+
pf = pq.ParquetFile(parquet_path)
|
| 30 |
+
n = 0
|
| 31 |
+
for batch in pf.iter_batches(batch_size=10000, columns=[col]):
|
| 32 |
+
for t in batch.column(0).to_pylist():
|
| 33 |
+
if not t:
|
| 34 |
+
continue
|
| 35 |
+
yield t
|
| 36 |
+
n += 1
|
| 37 |
+
if limit and n >= limit:
|
| 38 |
+
return
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def build(vocab_size):
|
| 42 |
+
tok = Tokenizer(models.BPE())
|
| 43 |
+
# ByteLevel with add_prefix_space=False matches the canonical pre_tokenizer/decoder.
|
| 44 |
+
tok.pre_tokenizer = pre_tokenizers.ByteLevel(add_prefix_space=False)
|
| 45 |
+
tok.decoder = decoders.ByteLevel()
|
| 46 |
+
trainer = trainers.BpeTrainer(
|
| 47 |
+
vocab_size=vocab_size,
|
| 48 |
+
special_tokens=SPECIAL,
|
| 49 |
+
initial_alphabet=pre_tokenizers.ByteLevel.alphabet(), # full 256-byte alphabet
|
| 50 |
+
show_progress=True,
|
| 51 |
+
)
|
| 52 |
+
return tok, trainer
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def main():
|
| 56 |
+
ap = argparse.ArgumentParser()
|
| 57 |
+
ap.add_argument("--corpus", default="C:/Projekty/datasets/build/en/en.parquet")
|
| 58 |
+
ap.add_argument("--out", default="tokenizer.json")
|
| 59 |
+
ap.add_argument("--col", default="text")
|
| 60 |
+
# 12288 = 12285 learned (256 bytes + 12029 merges) + 3 specials, as in canonical.
|
| 61 |
+
ap.add_argument("--vocab", type=int, default=12288)
|
| 62 |
+
ap.add_argument("--limit", type=int, default=None, help="cap #docs (smoke test)")
|
| 63 |
+
a = ap.parse_args()
|
| 64 |
+
|
| 65 |
+
tok, trainer = build(a.vocab)
|
| 66 |
+
tok.train_from_iterator(iter_text(a.corpus, a.col, a.limit), trainer=trainer)
|
| 67 |
+
tok.save(a.out)
|
| 68 |
+
print(f"saved {a.out} vocab_size={tok.get_vocab_size()} "
|
| 69 |
+
f"(specials={SPECIAL}, pre_tokenizer=ByteLevel)")
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
if __name__ == "__main__":
|
| 73 |
+
main()
|