Add lesson: Tokenizers (BPE, WordPiece, SentencePiece)

Phase 10, Lesson 01. Your LLM reads integers, not English. Covers
why subword tokenization won (word-level = infinite vocab, char-level
= too long), BPE step-by-step with worked examples, WordPiece
likelihood criterion, SentencePiece for multilingual, vocab size
tradeoffs with real numbers (GPT-2: 50,257, Llama 3: 128,256,
GPT-4o: 200,019). Pure Python BPE implementation with compression
ratio analysis and tiktoken comparison.
This commit is contained in:
Rohit Ghumare
2026-03-30 13:59:26 +01:00
parent b31a99b16a
commit 10a0faf3b8
3 changed files with 589 additions and 151 deletions
@@ -0,0 +1,227 @@
from collections import Counter
class CharTokenizer:
def encode(self, text):
return [ord(c) for c in text]
def decode(self, tokens):
return "".join(chr(t) for t in tokens)
class BPETokenizer:
def __init__(self):
self.merges = {}
self.vocab = {}
def _get_pairs(self, tokens):
pairs = Counter()
for i in range(len(tokens) - 1):
pairs[(tokens[i], tokens[i + 1])] += 1
return pairs
def _merge_pair(self, tokens, pair, new_token):
merged = []
i = 0
while i < len(tokens):
if i < len(tokens) - 1 and tokens[i] == pair[0] and tokens[i + 1] == pair[1]:
merged.append(new_token)
i += 2
else:
merged.append(tokens[i])
i += 1
return merged
def train(self, text, num_merges):
tokens = list(text.encode("utf-8"))
self.vocab = {i: bytes([i]) for i in range(256)}
for i in range(num_merges):
pairs = self._get_pairs(tokens)
if not pairs:
break
best_pair = max(pairs, key=pairs.get)
new_token = 256 + i
tokens = self._merge_pair(tokens, best_pair, new_token)
self.merges[best_pair] = new_token
self.vocab[new_token] = self.vocab[best_pair[0]] + self.vocab[best_pair[1]]
return self
def encode(self, text):
tokens = list(text.encode("utf-8"))
for pair, new_token in self.merges.items():
tokens = self._merge_pair(tokens, pair, new_token)
return tokens
def decode(self, tokens):
byte_sequence = b"".join(self.vocab[t] for t in tokens)
return byte_sequence.decode("utf-8", errors="replace")
def vocab_size(self):
return len(self.vocab)
def token_to_str(self, token_id):
return self.vocab.get(token_id, b"<?>").decode("utf-8", errors="replace")
def compression_ratio(tokenizer, text):
encoded = tokenizer.encode(text)
raw_bytes = len(text.encode("utf-8"))
return len(encoded) / raw_bytes
def vocabulary_stats(tokenizer, texts):
total_tokens = 0
total_words = 0
token_usage = Counter()
for text in texts:
encoded = tokenizer.encode(text)
total_tokens += len(encoded)
total_words += len(text.split())
for t in encoded:
token_usage[t] += 1
avg_tokens_per_word = total_tokens / total_words if total_words > 0 else 0
print(f"Vocabulary size: {tokenizer.vocab_size()}")
print(f"Avg tokens per word: {avg_tokens_per_word:.2f}")
print(f"Total unique tokens used: {len(token_usage)}")
print(f"\nTop 10 most used tokens:")
for token_id, count in token_usage.most_common(10):
display = tokenizer.token_to_str(token_id)
print(f" {token_id:4d}: '{display}' x{count}")
unused = tokenizer.vocab_size() - len(token_usage)
print(f"\nUnused tokens: {unused} out of {tokenizer.vocab_size()}")
def demo_char_tokenizer():
print("=" * 60)
print("STEP 1: Character-Level Tokenizer")
print("=" * 60)
ct = CharTokenizer()
texts = ["hello", "Hello, world!", "GPT-4"]
for text in texts:
encoded = ct.encode(text)
decoded = ct.decode(encoded)
print(f" '{text}' -> {encoded}")
print(f" Roundtrip: {'PASS' if decoded == text else 'FAIL'}")
print(f" Tokens: {len(encoded)}")
print()
def demo_bpe_training():
print("=" * 60)
print("STEP 2: BPE Training")
print("=" * 60)
corpus = (
"The cat sat on the mat. The cat ate the rat. "
"The dog sat on the log. The dog ate the frog. "
"Natural language processing is the study of how computers "
"understand and generate human language. "
"Tokenization is the first step in any NLP pipeline. "
"Language models read tokens, not words. "
"The tokenizer converts text into a sequence of integers. "
"Each integer maps to a subword in the vocabulary."
)
tokenizer = BPETokenizer()
tokenizer.train(corpus, num_merges=50)
print(f"\nVocabulary size after training: {tokenizer.vocab_size()}")
print(f"Number of merges learned: {len(tokenizer.merges)}")
return tokenizer, corpus
def demo_encode_decode(tokenizer):
print("\n" + "=" * 60)
print("STEP 3: Encode and Decode")
print("=" * 60)
test_sentences = [
"The cat sat on the mat.",
"Natural language processing",
"tokenization pipeline",
"unhappiness",
"The dog ate the frog.",
]
for sentence in test_sentences:
encoded = tokenizer.encode(sentence)
decoded = tokenizer.decode(encoded)
raw_bytes = len(sentence.encode("utf-8"))
ratio = len(encoded) / raw_bytes
roundtrip = "PASS" if decoded == sentence else "FAIL"
print(f"\n '{sentence}'")
print(f" Encoded: {encoded[:15]}{'...' if len(encoded) > 15 else ''}")
print(f" Tokens: {len(encoded)} (from {raw_bytes} bytes)")
print(f" Compression ratio: {ratio:.2f}")
print(f" Roundtrip: {roundtrip}")
def demo_tiktoken_comparison(tokenizer):
print("\n" + "=" * 60)
print("STEP 4: Compare with tiktoken")
print("=" * 60)
try:
import tiktoken
except ImportError:
print(" tiktoken not installed. Run: pip install tiktoken")
return
enc = tiktoken.get_encoding("cl100k_base")
texts = [
"The cat sat on the mat.",
"unhappiness",
"Hello, world!",
"def fibonacci(n): return n if n < 2 else fibonacci(n-1) + fibonacci(n-2)",
"Geschwindigkeitsbegrenzung",
]
for text in texts:
our_tokens = tokenizer.encode(text)
tk_tokens = enc.encode(text)
tk_pieces = [enc.decode([t]) for t in tk_tokens]
print(f"\n '{text}'")
print(f" Our BPE: {len(our_tokens)} tokens")
print(f" tiktoken: {len(tk_tokens)} tokens -> {tk_pieces}")
ratio = len(our_tokens) / len(tk_tokens) if len(tk_tokens) > 0 else 0
print(f" Ours / tiktoken: {ratio:.1f}x")
def demo_vocabulary_analysis(tokenizer, corpus):
print("\n" + "=" * 60)
print("STEP 5: Vocabulary Analysis")
print("=" * 60)
test_texts = [
corpus,
"The quick brown fox jumps over the lazy dog.",
"Machine learning is a subset of artificial intelligence.",
"Python is the most popular language for data science.",
]
vocabulary_stats(tokenizer, test_texts)
print(f"\nCompression ratios:")
for text in test_texts[:3]:
preview = text[:50] + "..." if len(text) > 50 else text
ratio = compression_ratio(tokenizer, text)
print(f" {ratio:.2f} -- '{preview}'")
if __name__ == "__main__":
demo_char_tokenizer()
tokenizer, corpus = demo_bpe_training()
demo_encode_decode(tokenizer)
demo_tiktoken_comparison(tokenizer)
demo_vocabulary_analysis(tokenizer, corpus)
@@ -1,153 +1,216 @@
# Tokenizers: BPE, WordPiece, SentencePiece
> The tokenizer is the front door of every language model - nothing gets in without passing through it first.
> Your LLM does not read English. It reads integers. The tokenizer decides whether those integers carry meaning or waste it.
**Type:** Build
**Languages:** Python, Rust
**Prerequisites:** Phase 5 (NLP Foundations)
**Languages:** Python
**Prerequisites:** Phase 05 (NLP Foundations)
**Time:** ~90 minutes
## The Problem
You feed a sentence into GPT. What does the model actually see?
Your LLM does not read English. It does not read any language. It reads numbers.
Not characters. Not words. Tokens.
The gap between "Hello, world!" and [15496, 11, 995, 0] is the tokenizer. Every word, every space, every punctuation mark must be converted into an integer before a model can process it. This conversion is not neutral. It bakes assumptions into the model that cannot be undone later.
The tokenizer decides how text gets sliced into pieces before a model ever touches it. That decision shapes everything downstream: vocabulary size, sequence length, out-of-vocabulary handling, multilingual support, arithmetic ability, even the cost of an API call (you pay per token).
Get this wrong and your model wastes capacity encoding common words with multiple tokens. "unfortunately" becomes four tokens instead of one. Your 128K context window just shrank by 75% for text heavy in multi-syllable words. Get it right and the same context window holds twice as much meaning. The difference between "this model handles code well" and "this model chokes on Python" often comes down to how the tokenizer was trained.
A bad tokenizer wastes context window on redundant subwords. A good one compresses common patterns and gracefully handles rare words. The difference between "this model understands code" and "this model chokes on variable names" often comes down to tokenizer design.
If you skip this, you will not understand why GPT tokenizes " New" and "York" separately, why BERT handles "[UNK]" tokens, or why some models burn through your context window twice as fast on non-English text.
Every API call you make to GPT-4 or Claude is priced per token. Every token your model generates costs compute. The fewer tokens required to represent an output, the faster the end-to-end inference. Tokenization is not preprocessing. It is architecture.
## The Concept
### Three Approaches to Splitting Text
### Three Approaches That Failed (and One That Won)
There are three obvious ways to convert text to numbers. Two of them do not work at scale.
**Word-level tokenization** splits on spaces and punctuation. "The cat sat" becomes ["The", "cat", "sat"]. Simple. But what about "tokenization"? Or "GPT-4o"? Or a German compound word like "Geschwindigkeitsbegrenzung"? Word-level requires a massive vocabulary to cover every word in every language. Miss a word and you get the dreaded `[UNK]` token -- the model's way of saying "I have no idea what this is." English alone has over a million word forms. Add code, URLs, scientific notation, and 100 other languages and you need an infinite vocabulary.
**Character-level tokenization** goes the other direction. "hello" becomes ["h", "e", "l", "l", "o"]. Vocabulary is tiny (a few hundred characters). No unknown tokens ever. But sequences become extremely long. A sentence that would be 10 word-level tokens becomes 50 character-level tokens. The model must learn that "t", "h", "e" together mean "the" -- burning attention capacity on something a human learns at age three.
**Subword tokenization** finds the sweet spot. Common words stay whole: "the" is one token. Rare words decompose into meaningful pieces: "unhappiness" becomes ["un", "happi", "ness"]. Vocabulary stays manageable (30K to 128K tokens). Sequences stay short. Unknown tokens essentially disappear because any word can be built from subword pieces.
Every modern LLM uses subword tokenization. GPT-2, GPT-4, BERT, Llama 3, Claude -- all of them. The question is which algorithm.
```mermaid
graph TD
A["Text: 'unhappiness'"] --> B{"Tokenization Strategy"}
B -->|Word-level| C["['unhappiness']\n1 token if in vocab\n[UNK] if not"]
B -->|Character-level| D["['u','n','h','a','p','p','i','n','e','s','s']\n11 tokens"]
B -->|Subword BPE| E["['un','happi','ness']\n3 tokens"]
style C fill:#ff6b6b,color:#fff
style D fill:#ffa500,color:#fff
style E fill:#51cf66,color:#fff
```
Character-level: "hello" -> ["h", "e", "l", "l", "o"]
Word-level: "hello world" -> ["hello", "world"]
Subword-level: "unhappiness" -> ["un", "happi", "ness"]
```
Each has tradeoffs:
| Approach | Vocabulary Size | Sequence Length | OOV Handling | Example |
|----------|----------------|-----------------|--------------|---------|
| Character | ~256 | Very long | None (all chars known) | GPT-1 early experiments |
| Word | 100K+ | Short | Poor (unknown words) | Classical NLP |
| Subword | 30K-100K | Medium | Good (decomposes unknowns) | GPT, BERT, LLaMA |
Subword tokenization won. Every modern LLM uses it. The question is which subword algorithm.
### BPE: Byte Pair Encoding
BPE starts with individual characters and repeatedly merges the most frequent adjacent pair. It is a greedy compression algorithm repurposed for tokenization.
BPE is a greedy compression algorithm repurposed for tokenization. The idea is simple enough to fit on an index card.
Here is how it works on a tiny corpus:
Start with individual characters. Count every adjacent pair in the training corpus. Merge the most frequent pair into a new token. Repeat until you reach your target vocabulary size.
Here is BPE running on a tiny corpus with the words "lower", "lowest", and "newest":
```
Corpus: "hug hug hug pug pug bug"
Corpus (with word frequencies):
"lower" x5
"lowest" x2
"newest" x6
Step 0 - Start with characters:
h u g h u g h u g p u g p u g b u g
Step 0 -- Start with characters:
l o w e r (x5)
l o w e s t (x2)
n e w e s t (x6)
Step 1 - Count all adjacent pairs:
(h,u): 3 (u,g): 6 (g, ): 5 ( ,h): 2
( ,p): 2 (p,u): 2 ( ,b): 1
Step 1 -- Count adjacent pairs:
(e,s): 8 (s,t): 8 (l,o): 7 (o,w): 7
(w,e): 13 (e,r): 5 (n,e): 6 ...
Step 2 - Merge most frequent pair (u,g) -> "ug":
h ug h ug h ug p ug p ug b ug
Step 2 -- Merge most frequent pair (w,e) -> "we":
l o we r (x5)
l o we s t (x2)
n e we s t (x6)
Step 3 - Recount pairs:
(h,ug): 3 (ug, ): 5 ( ,h): 2 ( ,p): 2
(p,ug): 2 ( ,b): 1
Step 3 -- Recount and merge (e,s) -> "es":
l o we r (x5)
l o we s t (x2) <- 'es' only forms from 'e'+'s', not 'we'+'s'
n e we s t (x6) <- wait, the 'e' before 'we' and 's' after 'we'
Step 4 - Merge most frequent (ug, ) -> "ug ":
h "ug " h "ug " h "ug " p "ug " p "ug " b ug
Actually tracking this precisely:
After "we" merge, remaining pairs:
(l,o): 7 (o,we): 7 (we,r): 5 (we,s): 8
(s,t): 8 (n,e): 6 (e,we): 6
Step 5 - Continue until vocabulary size target reached...
Step 3 -- Merge (we,s) -> "wes" or (s,t) -> "st" (tied at 8, pick first):
Merge (we,s) -> "wes":
l o we r (x5)
l o wes t (x2)
n e wes t (x6)
Step 4 -- Merge (wes,t) -> "west":
l o we r (x5)
l o west (x2)
n e west (x6)
...continue until target vocab size reached.
```
The merge table becomes your tokenizer. To encode new text, apply merges in the same order they were learned.
The merge table is the tokenizer. To encode new text, apply merges in the order they were learned. The training corpus determines which merges exist, and that choice permanently shapes what the model sees.
```
BPE Merge Process (ASCII diagram):
Input text: "unhappily"
Start: u n h a p p i l y
\/
Merge 1: un h a p p i l y (u+n -> un, if learned)
\ /
Merge 2: un h ap p i l y (a+p -> ap, if learned)
\ /
Merge 3: un h app i l y (ap+p -> app, if learned)
\ /
Merge 4: un happ i l y (h+app -> happ, if learned)
\ /
Merge 5: un happi l y (app+i -> appi, if learned)
\ /
Merge 6: un happi ly (l+y -> ly, if learned)
Result: ["un", "happi", "ly"]
```mermaid
graph LR
subgraph Training["BPE Training Loop"]
direction TB
T1["Start: character vocabulary"] --> T2["Count all adjacent pairs"]
T2 --> T3["Merge most frequent pair"]
T3 --> T4["Add merged token to vocab"]
T4 --> T5{"Reached target\nvocab size?"}
T5 -->|No| T2
T5 -->|Yes| T6["Done: save merge table"]
end
```
### Byte-Level BPE (GPT-2, GPT-3, GPT-4)
Standard BPE operates on Unicode characters. Byte-level BPE operates on raw bytes (0-255). This gives you a base vocabulary of exactly 256, handles any language or encoding, and never produces an unknown token.
GPT-2 introduced this. The trick: map each byte to a visible Unicode character so the vocabulary stays human-readable. The byte 0x20 (space) becomes "G", 0x41 ('A') stays 'A', and so on.
GPT-2 introduced this approach. The base vocabulary covers every possible byte. BPE merges build on top of that. OpenAI's tiktoken library implements byte-level BPE with these vocabulary sizes:
tiktoken (OpenAI's tokenizer library) uses byte-level BPE with a vocabulary of ~100K tokens for GPT-4.
- GPT-2: 50,257 tokens
- GPT-3.5/GPT-4: ~100,256 tokens (cl100k_base encoding)
- GPT-4o: 200,019 tokens (o200k_base encoding)
### WordPiece (BERT)
WordPiece is similar to BPE but picks merges differently. Instead of raw frequency, it maximizes the likelihood of the training data:
WordPiece looks similar to BPE but picks merges differently. Instead of raw frequency, it maximizes the likelihood of the training data:
```
BPE merge criterion: count(A, B)
WordPiece merge criterion: count(AB) / (count(A) * count(B))
```
WordPiece favors merges where the pair appears together more often than you would expect by chance. It also uses a "##" prefix for continuation tokens:
BPE asks: "Which pair appears most often?" WordPiece asks: "Which pair appears together more often than you would expect by chance?" This subtle difference produces different vocabularies. WordPiece favors merges where co-occurrence is surprising, not just frequent.
WordPiece also uses a "##" prefix for continuation subwords:
```
"unhappiness" -> ["un", "##happi", "##ness"]
"embedding" -> ["em", "##bed", "##ding"]
```
The "##" tells you this piece continues a previous token rather than starting a new word.
The "##" prefix tells you this piece continues a previous token. BERT uses WordPiece with a vocabulary of 30,522 tokens. Every BERT variant -- DistilBERT, RoBERTa's tokenizer is actually BPE, but BERT itself is WordPiece.
### SentencePiece
### SentencePiece (Llama, T5)
SentencePiece treats the input as a raw stream of Unicode characters, including whitespace. It does not require pre-tokenized words. This makes it language-agnostic - it works on Chinese, Japanese, Thai, and other languages where word boundaries are not marked by spaces.
SentencePiece treats the input as a raw stream of Unicode characters, including whitespace. No pre-tokenization step. No language-specific rules about word boundaries. This makes it genuinely language-agnostic -- it works on Chinese, Japanese, Thai, and other languages where spaces do not separate words.
SentencePiece supports both BPE and Unigram algorithms. LLaMA, T5, and many multilingual models use SentencePiece.
SentencePiece supports two algorithms:
- **BPE mode**: same merge logic as standard BPE, applied to raw character sequences
- **Unigram mode**: starts with a large vocabulary and iteratively removes tokens that least affect the overall likelihood. The reverse of BPE -- prune instead of merge.
The Unigram approach works in reverse compared to BPE: start with a large vocabulary and iteratively remove tokens that least affect the overall likelihood.
Llama 3 uses SentencePiece with a vocabulary of 128,256 tokens. T5 uses SentencePiece Unigram with 32,000 tokens.
### How Tokenizer Choice Affects the Model
### Vocabulary Size Tradeoffs
The tokenizer is not neutral. It bakes in assumptions:
This is a real engineering decision with measurable consequences.
**Vocabulary size tradeoff:**
- Larger vocabulary (100K+): shorter sequences, more parameters in the embedding layer
- Smaller vocabulary (30K): longer sequences, smaller embedding layer, better generalization to rare words
```mermaid
graph LR
subgraph Small["Small Vocab (32K)\ne.g., BERT, T5"]
S1["More tokens per text"]
S2["Longer sequences"]
S3["Smaller embedding matrix"]
S4["Better rare-word handling"]
end
subgraph Large["Large Vocab (128K+)\ne.g., Llama 3, GPT-4o"]
L1["Fewer tokens per text"]
L2["Shorter sequences"]
L3["Larger embedding matrix"]
L4["Faster inference"]
end
```
**Fertility (tokens per word):**
- English text in GPT-4: ~1.3 tokens per word
- Korean text in GPT-4: ~2-3 tokens per word
- Code: highly variable, depends on training data
Concrete numbers. For a 128K vocabulary with 4,096-dimensional embeddings, the embedding matrix alone is 128,000 x 4,096 = 524 million parameters. For a 32K vocabulary, it is 131 million parameters. That is a 400M parameter difference from the tokenizer choice alone.
**Downstream effects:**
- Arithmetic: "1234" tokenized as ["123", "4"] vs ["1", "234"] changes whether the model can learn digit-level operations
- Code: a tokenizer trained mostly on English text wastes tokens on Python indentation
- Multilingual: models that tokenize non-English text into many small pieces effectively have a shorter context window for those languages
But larger vocabularies compress text more aggressively. The same English paragraph that takes 100 tokens with a 32K vocabulary might take 70 tokens with a 128K vocabulary. That means 30% fewer forward passes during generation. For a model serving millions of requests, that is a direct reduction in compute cost.
The trend is clear: vocabulary sizes are growing. GPT-2 used 50,257. GPT-4 uses ~100K. Llama 3 uses 128K. GPT-4o uses 200K.
| Model | Vocab Size | Tokenizer Type | Avg Tokens per English Word |
|-------|-----------|----------------|---------------------------|
| BERT | 30,522 | WordPiece | ~1.4 |
| GPT-2 | 50,257 | Byte-level BPE | ~1.3 |
| Llama 2 | 32,000 | SentencePiece BPE | ~1.4 |
| GPT-4 | ~100,256 | Byte-level BPE | ~1.2 |
| Llama 3 | 128,256 | SentencePiece BPE | ~1.1 |
| GPT-4o | 200,019 | Byte-level BPE | ~1.0 |
### The Multilingual Tax
Tokenizers trained primarily on English are brutal to other languages. Korean text in GPT-2's tokenizer averages 2-3 tokens per word. Chinese can be worse. This means a Korean user effectively has a context window that is half the size of an English user's -- paying the same price for less information density.
This is why Llama 3 quadrupled its vocabulary from 32K to 128K. More tokens dedicated to non-English scripts means fairer compression across languages.
## Build It
### Step 1: Basic BPE Tokenizer
### Step 1: Character-Level Tokenizer
We build a complete BPE tokenizer from scratch. The training loop counts pairs, finds the most frequent, merges it, and records the merge rule.
Start at the foundation. A character-level tokenizer maps each character to its Unicode code point. No training needed. No unknown tokens. Just a direct mapping.
```python
class CharTokenizer:
def encode(self, text):
return [ord(c) for c in text]
def decode(self, tokens):
return "".join(chr(t) for t in tokens)
```
"hello" becomes [104, 101, 108, 108, 111]. Every character is its own token. This is the baseline we improve on.
### Step 2: BPE Tokenizer from Scratch
The real implementation. We train on raw bytes (like GPT-2), count pairs, merge the most frequent, and record every merge in order. The merge table is the tokenizer.
```python
from collections import Counter
@@ -188,8 +251,8 @@ class BPETokenizer:
tokens = self._merge_pair(tokens, best_pair, new_token)
self.merges[best_pair] = new_token
self.vocab[new_token] = self.vocab[best_pair[0]] + self.vocab[best_pair[1]]
merged_str = self.vocab[new_token]
print(f"Merge {i + 1}: {best_pair} -> {new_token} = {merged_str}")
return self
def encode(self, text):
tokens = list(text.encode("utf-8"))
@@ -202,73 +265,124 @@ class BPETokenizer:
return byte_sequence.decode("utf-8", errors="replace")
```
### Step 2: Train and Test
The training loop is the core of BPE: count pairs, merge the winner, repeat. Each merge reduces the total token count. After `num_merges` rounds, the vocabulary grows from 256 (base bytes) to 256 + num_merges.
Encoding applies merges in the exact order they were learned. This matters. If merge 1 created "th" and merge 5 created "the", encoding must apply merge 1 first so that "the" can form from "th" + "e" in merge 5.
Decoding is the inverse: look up each token ID in the vocabulary, concatenate the bytes, decode to UTF-8.
### Step 3: Encode and Decode Roundtrip
```python
corpus = """The cat sat on the mat. The cat ate the rat.
The dog sat on the log. The dog ate the frog.
Natural language processing is the study of how computers
understand and generate human language."""
corpus = (
"The cat sat on the mat. The cat ate the rat. "
"The dog sat on the log. The dog ate the frog. "
"Natural language processing is the study of how computers "
"understand and generate human language. "
"Tokenization is the first step in any NLP pipeline."
)
tokenizer = BPETokenizer()
tokenizer.train(corpus, num_merges=30)
tokenizer.train(corpus, num_merges=40)
test = "The cat sat on the mat."
encoded = tokenizer.encode(test)
decoded = tokenizer.decode(encoded)
test_sentences = [
"The cat sat on the mat.",
"Natural language processing",
"tokenization pipeline",
"unhappiness",
]
print(f"\nOriginal: {test}")
print(f"Encoded: {encoded}")
print(f"Decoded: {decoded}")
print(f"Tokens: {len(encoded)} (from {len(test.encode('utf-8'))} bytes)")
for sentence in test_sentences:
encoded = tokenizer.encode(sentence)
decoded = tokenizer.decode(encoded)
raw_bytes = len(sentence.encode("utf-8"))
ratio = len(encoded) / raw_bytes
print(f"'{sentence}'")
print(f" Tokens: {len(encoded)} (from {raw_bytes} bytes) -- ratio: {ratio:.2f}")
print(f" Roundtrip: {'PASS' if decoded == sentence else 'FAIL'}")
```
### Step 3: Compare With tiktoken
The compression ratio tells you how effective the tokenizer is. A ratio of 0.50 means the tokenizer compressed the text to half as many tokens as raw bytes. Lower is better. On the training corpus, the ratio will be good. On out-of-distribution text like "unhappiness" (which does not appear in the corpus), the ratio will be worse -- the tokenizer falls back to character-level encoding for unseen patterns.
### Step 4: Compare with tiktoken
```python
import tiktoken
enc = tiktoken.get_encoding("cl100k_base")
text = "The cat sat on the mat."
tokens = enc.encode(text)
print(f"tiktoken tokens: {tokens}")
print(f"tiktoken decoded: {[enc.decode([t]) for t in tokens]}")
print(f"Token count: {len(tokens)}")
texts = [
"The cat sat on the mat.",
"unhappiness",
"Hello, world!",
"def fibonacci(n): return n if n < 2 else fibonacci(n-1) + fibonacci(n-2)",
"Geschwindigkeitsbegrenzung",
]
text2 = "unhappiness"
tokens2 = enc.encode(text2)
print(f"\n'{text2}' -> {[enc.decode([t]) for t in tokens2]}")
print(f"Token count: {len(tokens2)}")
for text in texts:
our_tokens = tokenizer.encode(text)
tiktoken_tokens = enc.encode(text)
tiktoken_pieces = [enc.decode([t]) for t in tiktoken_tokens]
print(f"'{text}'")
print(f" Our BPE: {len(our_tokens)} tokens")
print(f" tiktoken: {len(tiktoken_tokens)} tokens -> {tiktoken_pieces}")
```
tiktoken uses the same BPE algorithm, but trained on a massive corpus with 100K merges. The merge table is what makes it powerful, not the algorithm itself.
tiktoken uses the exact same algorithm but trained on hundreds of gigabytes of text with 100,000 merges. The algorithm is identical. The difference is the training data and the number of merges. Your tokenizer trained on a paragraph with 40 merges cannot compete with tiktoken's 100K merges on a massive corpus. But the mechanism is the same.
### Step 5: Vocabulary Analysis
```python
def analyze_vocabulary(tokenizer, test_texts):
total_tokens = 0
total_chars = 0
token_usage = Counter()
for text in test_texts:
encoded = tokenizer.encode(text)
total_tokens += len(encoded)
total_chars += len(text)
for t in encoded:
token_usage[t] += 1
print(f"Vocabulary size: {len(tokenizer.vocab)}")
print(f"Total tokens across all texts: {total_tokens}")
print(f"Total characters: {total_chars}")
print(f"Avg tokens per character: {total_tokens / total_chars:.2f}")
print(f"\nMost used tokens:")
for token_id, count in token_usage.most_common(10):
token_bytes = tokenizer.vocab[token_id]
display = token_bytes.decode("utf-8", errors="replace")
print(f" Token {token_id:4d}: '{display}' (used {count} times)")
unused = [t for t in tokenizer.vocab if t not in token_usage]
print(f"\nUnused tokens: {len(unused)} out of {len(tokenizer.vocab)}")
```
This reveals the Zipf distribution in your vocabulary. A few tokens dominate (spaces, "the", "e"). Most tokens are rarely used. Production tokenizers optimize for this distribution -- common patterns get short token IDs, rare patterns get longer representations.
## Use It
### SentencePiece
Your scratch BPE works. Now see what production tools look like.
### tiktoken (OpenAI)
```python
import sentencepiece as spm
import tiktoken
spm.SentencePieceTrainer.train(
input="corpus.txt",
model_prefix="my_tokenizer",
vocab_size=1000,
model_type="bpe"
)
enc = tiktoken.get_encoding("cl100k_base")
sp = spm.SentencePieceProcessor()
sp.load("my_tokenizer.model")
tokens = sp.encode("The cat sat on the mat.", out_type=str)
print(tokens)
ids = sp.encode("The cat sat on the mat.")
print(ids)
print(sp.decode(ids))
text = "Tokenizers convert text to integers"
tokens = enc.encode(text)
print(f"Tokens: {tokens}")
print(f"Pieces: {[enc.decode([t]) for t in tokens]}")
print(f"Roundtrip: {enc.decode(tokens)}")
```
### Hugging Face Tokenizers
tiktoken is written in Rust with Python bindings. It encodes millions of tokens per second. Same BPE algorithm, industrial-strength implementation.
### Hugging Face tokenizers
```python
from tokenizers import Tokenizer
@@ -283,45 +397,68 @@ trainer = BpeTrainer(vocab_size=1000, special_tokens=["<pad>", "<eos>", "<unk>"]
tokenizer.train(["corpus.txt"], trainer)
output = tokenizer.encode("The cat sat on the mat.")
print(output.tokens)
print(output.ids)
print(f"Tokens: {output.tokens}")
print(f"IDs: {output.ids}")
```
The Hugging Face `tokenizers` library is written in Rust under the hood. It trains BPE on gigabyte-scale corpora in seconds.
The Hugging Face tokenizers library is also Rust under the hood. It trains BPE on gigabyte-scale corpora in seconds. This is what you use when training your own model.
### Rust for Production Tokenization
### Loading Llama's Tokenizer
When you need to tokenize millions of documents for pre-training, Python becomes the bottleneck. The Rust implementation in `code/bpe.rs` shows how the same algorithm runs 10-50x faster with zero-copy byte handling.
```python
from transformers import AutoTokenizer
tokenizer = AutoTokenizer.from_pretrained("meta-llama/Llama-3.1-8B")
text = "Tokenizers are the unsung heroes of LLMs"
tokens = tokenizer.encode(text)
print(f"Token IDs: {tokens}")
print(f"Tokens: {tokenizer.convert_ids_to_tokens(tokens)}")
print(f"Vocab size: {tokenizer.vocab_size}")
multilingual = ["Hello world", "Hola mundo", "Bonjour le monde"]
for text in multilingual:
ids = tokenizer.encode(text)
print(f"'{text}' -> {len(ids)} tokens")
```
Llama 3's 128K vocabulary compresses non-English text significantly better than GPT-2's 50K vocabulary. You can verify this yourself -- encode the same sentence in multiple languages and count the tokens.
## Ship It
This lesson produces a skill for choosing and building tokenizers in LLM projects. See `outputs/skill-tokenizer.md`.
This lesson produces `outputs/prompt-tokenizer-analyzer.md` -- a reusable prompt that analyzes tokenization efficiency for any text and model combination. Feed it a text sample and it tells you which model's tokenizer handles it best.
## Exercises
1. **Easy:** Modify the BPE tokenizer to print the vocabulary at each merge step. Observe how common English words get assembled piece by piece.
2. **Medium:** Add special tokens (`<pad>`, `<eos>`, `<unk>`) to the BPE tokenizer. Implement pre-tokenization that splits on whitespace before running BPE.
3. **Hard:** Implement the WordPiece merge criterion (likelihood-based instead of frequency-based). Compare the vocabularies produced by BPE vs WordPiece on the same corpus.
1. Modify the BPE tokenizer to print the vocabulary at each merge step. Watch how "t" + "h" becomes "th", then "th" + "e" becomes "the". Track how common English words get assembled piece by piece.
2. Add special tokens (`<pad>`, `<eos>`, `<unk>`) to the BPE tokenizer. Assign them IDs 0, 1, 2 and shift all other tokens accordingly. Implement a pre-tokenization step that splits on whitespace before running BPE.
3. Implement the WordPiece merge criterion (likelihood ratio instead of frequency). Train both BPE and WordPiece on the same corpus with the same number of merges. Compare the resulting vocabularies -- which one produces more linguistically meaningful subwords?
4. Build a multilingual tokenizer efficiency benchmark. Take 10 sentences in English, Spanish, Chinese, Korean, and Arabic. Tokenize each with tiktoken (cl100k_base) and measure the average tokens per character. Quantify the "multilingual tax" for each language.
5. Train your BPE tokenizer on a larger corpus (download a Wikipedia article). Tune the number of merges to achieve a compression ratio within 10% of tiktoken on that same text. This forces you to understand the relationship between corpus size, merge count, and compression quality.
## Key Terms
| Term | What people say | What it actually means |
|------|----------------|----------------------|
| Token | "A word" | A unit in the model's vocabulary - could be a character, subword, word, or multi-word chunk |
| BPE | "Some compression thing" | Byte Pair Encoding - iteratively merge the most frequent adjacent pair of tokens |
| WordPiece | "BERT's tokenizer" | Like BPE but merges maximize training data likelihood instead of raw frequency |
| SentencePiece | "A tokenizer library" | A language-agnostic tokenizer that operates on raw Unicode, supporting BPE and Unigram algorithms |
| Vocabulary size | "How many words it knows" | The total number of unique tokens the model can represent - typically 30K to 100K |
| Fertility | "Not a tokenizer term" | Average number of tokens per word - measures tokenizer efficiency across languages |
| Byte-level BPE | "GPT's tokenizer" | BPE operating on raw bytes (0-255) instead of Unicode characters - guarantees no unknown tokens |
| Merge table | "The tokenizer file" | Ordered list of pair merges learned during training - this IS the tokenizer |
| Token | "A word" | A unit in the model's vocabulary -- could be a character, subword, word, or multi-word chunk |
| BPE | "Some compression thing" | Byte Pair Encoding -- iteratively merge the most frequent adjacent pair of tokens until the target vocabulary size is reached |
| WordPiece | "BERT's tokenizer" | Like BPE but merges maximize the likelihood ratio count(AB)/(count(A)*count(B)) instead of raw frequency |
| SentencePiece | "A tokenizer library" | A language-agnostic tokenizer that operates on raw Unicode without pre-tokenization, supporting BPE and Unigram algorithms |
| Vocabulary size | "How many words it knows" | The total number of unique tokens: GPT-2 has 50,257, BERT has 30,522, Llama 3 has 128,256 |
| Fertility | "Not a tokenizer term" | Average number of tokens per word -- measures tokenizer efficiency across languages (1.0 is perfect, 3.0 means the model works three times harder) |
| Byte-level BPE | "GPT's tokenizer" | BPE operating on raw bytes (0-255) instead of Unicode characters, guaranteeing no unknown tokens for any input |
| Merge table | "The tokenizer file" | Ordered list of pair merges learned during training -- this IS the tokenizer, and order matters |
| Pre-tokenization | "Splitting on spaces" | Rules applied before subword tokenization: whitespace splitting, digit separation, punctuation handling |
| tiktoken | "OpenAI's tokenizer" | OpenAI's fast BPE implementation used by GPT-3.5/4, with ~100K vocabulary |
| Compression ratio | "How efficient the tokenizer is" | Tokens produced divided by input bytes -- lower means better compression and faster inference |
## Further Reading
- [Sennrich et al., 2016 - Neural Machine Translation of Rare Words with Subword Units](https://arxiv.org/abs/1508.07909) - the paper that introduced BPE for NLP
- [Kudo & Richardson, 2018 - SentencePiece](https://arxiv.org/abs/1808.06226) - language-agnostic subword tokenization
- [Hugging Face Tokenizers documentation](https://huggingface.co/docs/tokenizers) - production-grade tokenizer training
- [Andrej Karpathy's minbpe](https://github.com/karpathy/minbpe) - minimal BPE implementation for education
- [tiktoken source code](https://github.com/openai/tiktoken) - OpenAI's Rust+Python BPE tokenizer
- [Sennrich et al., 2016 -- "Neural Machine Translation of Rare Words with Subword Units"](https://arxiv.org/abs/1508.07909) -- the paper that introduced BPE for NLP, turning a 1994 compression algorithm into the foundation of modern tokenization
- [Kudo & Richardson, 2018 -- "SentencePiece: A simple and language independent subword tokenizer"](https://arxiv.org/abs/1808.06226) -- language-agnostic tokenization that made multilingual models practical
- [OpenAI tiktoken repository](https://github.com/openai/tiktoken) -- production BPE implementation in Rust with Python bindings, used by GPT-3.5/4/4o
- [Andrej Karpathy's minbpe](https://github.com/karpathy/minbpe) -- minimal BPE implementation for education, the cleanest reference for understanding the algorithm
- [Hugging Face Tokenizers documentation](https://huggingface.co/docs/tokenizers) -- production-grade tokenizer training with Rust performance
@@ -0,0 +1,74 @@
---
name: prompt-tokenizer-analyzer
description: Analyze tokenization efficiency for a given text across different models and tokenizer types
phase: 10
lesson: 01
---
You are a tokenization efficiency analyst. I will give you a text sample and you will analyze how different tokenizers handle it, identify inefficiencies, and recommend the best tokenizer for the use case.
## Analysis Protocol
When I provide a text sample, follow this sequence:
### 1. Characterize the Text
Determine the text properties that affect tokenization:
- **Language distribution**: what percentage is English vs other languages vs code vs numbers vs special characters
- **Domain**: general text, code, scientific notation, URLs, structured data
- **Vocabulary profile**: common words vs domain-specific terms vs rare words
- **Script types**: Latin, CJK, Cyrillic, Arabic, emoji, mixed
### 2. Estimate Token Counts
For each major tokenizer, estimate the token count and explain why:
- **GPT-4 (cl100k_base)**: byte-level BPE, ~100K vocab
- **GPT-4o (o200k_base)**: byte-level BPE, ~200K vocab
- **BERT (WordPiece)**: 30K vocab, uses ## continuation tokens
- **Llama 3 (SentencePiece)**: 128K vocab, trained on multilingual data
Provide the estimate as tokens per 100 characters of input.
### 3. Identify Tokenization Inefficiencies
Flag specific patterns that waste tokens:
- Words that split into 3+ tokens (high fertility)
- Repeated subwords that could be single tokens with a larger vocabulary
- Whitespace or formatting consuming unnecessary tokens
- Numbers tokenized inconsistently (e.g., "1234" as ["123", "4"] vs ["1", "234"])
- Non-English text paying a "multilingual tax" (2x+ more tokens than English equivalent)
### 4. Calculate the Cost Impact
For each tokenizer, estimate:
- **Context utilization**: what percentage of a 128K context window this text would consume
- **Generation cost**: relative cost if this text were generated (more tokens = more cost)
- **Inference speed**: relative speed impact (more tokens = slower generation)
### 5. Recommend
Based on the analysis:
- Which tokenizer is most efficient for this specific text
- Whether a custom tokenizer trained on domain data would help
- Specific vocabulary size recommendation if training from scratch
- Pre-tokenization rules that would improve efficiency (digit splitting, whitespace handling)
## Input Format
Provide:
- The text sample (or a representative excerpt)
- The intended use case (training data, inference input, generation output)
- Any constraints (max context length, cost budget, latency requirements)
## Output Format
1. **Text Profile**: one-paragraph characterization of the text
2. **Token Count Estimates**: table with tokenizer name, estimated tokens, and tokens per 100 chars
3. **Inefficiency Report**: bulleted list of specific tokenization problems found
4. **Cost Analysis**: table showing context utilization, relative cost, and speed for each tokenizer
5. **Recommendation**: which tokenizer to use and why, with specific configuration if training custom