reading this record
code is folded so the argument and saved outputs come first. expand any cell to inspect the implementation. saved outputs are historical evidence from the source notebook, not a new run or independent verification.
inspect code · cell 01
In [1]:
import sys
from pathlib import Path
repo_root = Path.cwd()
if not (repo_root / "data").exists():
repo_root = repo_root.parent
sys.path.insert(0, str(repo_root))
from src.tokenizer import (
bpe_decode,
bpe_encode,
decode_unicode_tokens,
encode_unicode_string,
perform_merges,
save_tokenizer,
)
# Initial text
text = "Hello, world! This is a sample text for testing."
torture_text = "Hello, world! café پاکستان 日本語 🤖 ∇²ψ" # Cursed unicode string as a stress test
Implementing a naive tokenizer¶
inspect code · cell 02
In [2]:
chars = sorted(set(text))
char_to_index = {char: index for index, char in enumerate(chars)}
index_to_char = {index: char for index, char in enumerate(chars)}
def naive_encode(text):
# This function encodes the input text into a list of tokens.
return [char_to_index[char] for char in text]
def naive_decode(tokens):
# This function decodes the list of tokens back into the original text.
return "".join(index_to_char[token] for token in tokens)
encoded_text = naive_encode("Hello world! This is a test.")
print("Encoded:", encoded_text)
decoded_text = naive_decode(encoded_text)
print("Decoded:", decoded_text)
Byte-level tokenizer¶
inspect code · cell 03
In [3]:
# Unicode string → UTF-8 bytes → integer tokens
encoded_torture_text = encode_unicode_string(torture_text)
print("Encoded torture text:", encoded_torture_text)
decoded_torture_text = decode_unicode_tokens(encoded_torture_text)
print("Decoded torture text:", decoded_torture_text)
BPE (byte-pair encoding)¶
inspect code · cell 04
In [4]:
target_vocab_size = 512
tokens = encode_unicode_string(torture_text)
tokens, vocab, merges = perform_merges(tokens, target_vocab_size)
print("Final tokens:", tokens)
inspect code · cell 05
In [5]:
# Tiny Shakespeare corpus
corpus = (repo_root / "data" / "tiny_shakespeare.txt").read_text(encoding="utf-8")
tokens = encode_unicode_string(
corpus[:100_000]
) # Limit to first 100,000 characters for demonstration
fin_tokens, vocab, merges = perform_merges(tokens, target_vocab_size)
save_tokenizer(repo_root / "data" / "tiny_shakespeare_bpe.json", vocab, merges)
print(f"Encoded corpus into {len(fin_tokens):,} tokens")
print("Saved tokenizer to data/tiny_shakespeare_bpe.json")
inspect code · cell 06
In [6]:
test = "To be, or not to be 🤖"
encoded = bpe_encode(test, vocab, merges)
decoded = bpe_decode(encoded, vocab)
print("Original:", test)
print("Encoded:", encoded)
print("Decoded:", decoded)
assert test == decoded, "Decoded text does not match the original!"
inspect code · cell 07
In [7]:
tests = [
"hello world",
"To be, or not to be",
"aaaaaaaaaaaa",
"hello\nworld",
"🐸",
"پاکستان",
"hello 🐸 پاکستان",
"",
]
for test in tests:
encoded = bpe_encode(test, vocab, merges)
decoded = bpe_decode(encoded, vocab)
print("Original:", test)
print("Encoded:", encoded)
print("Decoded:", decoded)
assert test == decoded, "Decoded text does not match the original!"