Quazim0t0 commited on
Commit
e9a4fdd
·
verified ·
1 Parent(s): 15d845e

Upload special_tokens.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. special_tokens.py +85 -0
special_tokens.py ADDED
@@ -0,0 +1,85 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ special_tokens.py -- central, *append-only* registry of special tokens for the
3
+ SpikeWhale length-max tokenizer.
4
+
5
+ WHY THIS FILE EXISTS
6
+ --------------------
7
+ The base vocab (tokenizer.json) is 16384 contiguous ids:
8
+ 0..3 -> <pad> <unk> <bos> <eos>
9
+ 4..259 -> the 256 raw bytes
10
+ 260.. -> learned byte-merges
11
+ Adding tokens "without breaking the model" has exactly one rule:
12
+
13
+ ***APPEND ONLY. NEVER REORDER OR REMOVE AN EXISTING ID.***
14
+
15
+ Every existing id keeps pointing at the same embedding row and the same logit
16
+ column, so the model's behaviour on already-seen tokens is bit-for-bit
17
+ unchanged. New tokens are appended at ids >= 16384 and their embedding / lm_head
18
+ / mtp rows are freshly initialised (near-zero contribution) so they are
19
+ no-ops until you train them.
20
+
21
+ To stay tensor-core friendly the final vocab is padded up to a multiple of
22
+ `VOCAB_MULTIPLE` (128) with `<|reserved_N|>` slots. Those reserves let you name
23
+ *future* tokens later by editing the registry WITHOUT another model resize, as
24
+ long as the total stays <= the padded size.
25
+
26
+ HOW TO ADD MORE LATER
27
+ ---------------------
28
+ Append new names to NAMED_SPECIAL_TOKENS (at the END), then either:
29
+ * if you still have <|reserved_*|> slots free, just rename a reserved id in
30
+ tokenizer.json (no model change needed), or
31
+ * re-run add_special_tokens.py to grow + re-pad the vocab (model resized).
32
+ """
33
+
34
+ # Tensor-core / matmul friendly vocab alignment. 16384 is already 128*128.
35
+ VOCAB_MULTIPLE = 128
36
+
37
+ # ---------------------------------------------------------------------------
38
+ # The universal named set. ORDER IS PERMANENT -- append only, never reorder.
39
+ # Mixing the common conventions so the same model can do chat, reasoning,
40
+ # agentic tool use, and code infilling.
41
+ # ---------------------------------------------------------------------------
42
+ NAMED_SPECIAL_TOKENS = [
43
+ # ChatML turn framing
44
+ "<|im_start|>",
45
+ "<|im_end|>",
46
+ # Reasoning / scratchpad
47
+ "<think>",
48
+ "</think>",
49
+ # Explicit solution block
50
+ "<begin_solution>",
51
+ "<end_solution>",
52
+ # Agentic tool calling
53
+ "<tool_call>",
54
+ "</tool_call>",
55
+ "<tool_response>",
56
+ "</tool_response>",
57
+ # Role markers (usable standalone or inside an im_start header)
58
+ "<|system|>",
59
+ "<|user|>",
60
+ "<|assistant|>",
61
+ # Fill-in-the-middle (code)
62
+ "<|fim_prefix|>",
63
+ "<|fim_middle|>",
64
+ "<|fim_suffix|>",
65
+ # Generic document separator
66
+ "<|endoftext|>",
67
+ ]
68
+
69
+
70
+ def build_special_token_list(base_vocab_size: int,
71
+ multiple: int = VOCAB_MULTIPLE):
72
+ """
73
+ Return the ordered list of tokens to APPEND after `base_vocab_size`:
74
+ the named set followed by enough <|reserved_N|> slots to pad the final
75
+ vocab size up to the next multiple of `multiple`.
76
+
77
+ The returned list's element i gets id (base_vocab_size + i).
78
+ """
79
+ tokens = list(NAMED_SPECIAL_TOKENS)
80
+ target = base_vocab_size + len(tokens)
81
+ # round up to the next multiple (or stay put if already aligned)
82
+ padded = ((target + multiple - 1) // multiple) * multiple
83
+ n_reserved = padded - target
84
+ tokens += [f"<|reserved_{i}|>" for i in range(n_reserved)]
85
+ return tokens