Maggio33 commited on
Commit
2a8d44e
·
verified ·
1 Parent(s): 6a0572b

Upload train_tokenizer.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. train_tokenizer.py +73 -0
train_tokenizer.py ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """train_tokenizer.py - GoLLeM-v5 (EN) tokenizer training recipe.
3
+
4
+ Reproduces the canonical `tokenizer.json` shipped in this repo:
5
+ - byte-level BPE (GPT-2 lineage), pre-tokenizer + decoder = ByteLevel
6
+ - model vocab 12285 (256 byte alphabet + 12029 learned merges)
7
+ - 3 special tokens appended: <|endoftext|> <|im_start|> <|im_end|>
8
+ (=> 12288 effective ids; models pad the embedding to 12288)
9
+ - trained on the minimal-en-corpus (FineWeb-Edu EN broad mix; `en.parquet`,
10
+ the same source uploaded as SlayerLab/minimal-en-corpus-5b).
11
+
12
+ The shipped `tokenizer.json` remains the canonical / authoritative artifact:
13
+ a fresh run reproduces a functionally-equivalent tokenizer, but exact merge
14
+ order depends on the corpus snapshot/order, so byte-identity is not guaranteed.
15
+ Use this script to audit the method; use `tokenizer.json` for exact parity.
16
+
17
+ Usage:
18
+ python train_tokenizer.py --corpus en.parquet --out tokenizer.json
19
+ python train_tokenizer.py --limit 50000 --out tok_smoke.json # quick recipe check
20
+ """
21
+ import argparse
22
+ import pyarrow.parquet as pq
23
+ from tokenizers import Tokenizer, models, trainers, pre_tokenizers, decoders
24
+
25
+ SPECIAL = ["<|endoftext|>", "<|im_start|>", "<|im_end|>"]
26
+
27
+
28
+ def iter_text(parquet_path, col="text", limit=None):
29
+ pf = pq.ParquetFile(parquet_path)
30
+ n = 0
31
+ for batch in pf.iter_batches(batch_size=10000, columns=[col]):
32
+ for t in batch.column(0).to_pylist():
33
+ if not t:
34
+ continue
35
+ yield t
36
+ n += 1
37
+ if limit and n >= limit:
38
+ return
39
+
40
+
41
+ def build(vocab_size):
42
+ tok = Tokenizer(models.BPE())
43
+ # ByteLevel with add_prefix_space=False matches the canonical pre_tokenizer/decoder.
44
+ tok.pre_tokenizer = pre_tokenizers.ByteLevel(add_prefix_space=False)
45
+ tok.decoder = decoders.ByteLevel()
46
+ trainer = trainers.BpeTrainer(
47
+ vocab_size=vocab_size,
48
+ special_tokens=SPECIAL,
49
+ initial_alphabet=pre_tokenizers.ByteLevel.alphabet(), # full 256-byte alphabet
50
+ show_progress=True,
51
+ )
52
+ return tok, trainer
53
+
54
+
55
+ def main():
56
+ ap = argparse.ArgumentParser()
57
+ ap.add_argument("--corpus", default="C:/Projekty/datasets/build/en/en.parquet")
58
+ ap.add_argument("--out", default="tokenizer.json")
59
+ ap.add_argument("--col", default="text")
60
+ # 12288 = 12285 learned (256 bytes + 12029 merges) + 3 specials, as in canonical.
61
+ ap.add_argument("--vocab", type=int, default=12288)
62
+ ap.add_argument("--limit", type=int, default=None, help="cap #docs (smoke test)")
63
+ a = ap.parse_args()
64
+
65
+ tok, trainer = build(a.vocab)
66
+ tok.train_from_iterator(iter_text(a.corpus, a.col, a.limit), trainer=trainer)
67
+ tok.save(a.out)
68
+ print(f"saved {a.out} vocab_size={tok.get_vocab_size()} "
69
+ f"(specials={SPECIAL}, pre_tokenizer=ByteLevel)")
70
+
71
+
72
+ if __name__ == "__main__":
73
+ main()