{
 "cells": [
  {
   "cell_type": "code",
   "execution_count": 1,
   "id": "5702dffb",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:31.177376Z",
     "iopub.status.busy": "2026-08-23T00:04:31.177157Z",
     "iopub.status.idle": "2026-08-23T00:04:31.189071Z",
     "shell.execute_reply": "2026-08-23T00:04:31.188140Z"
    }
   },
   "outputs": [],
   "source": [
    "import sys\n",
    "from pathlib import Path\n",
    "\n",
    "repo_root = Path.cwd()\n",
    "if not (repo_root / \"data\").exists():\n",
    "    repo_root = repo_root.parent\n",
    "sys.path.insert(0, str(repo_root))\n",
    "\n",
    "from src.tokenizer import (\n",
    "    bpe_decode,\n",
    "    bpe_encode,\n",
    "    decode_unicode_tokens,\n",
    "    encode_unicode_string,\n",
    "    perform_merges,\n",
    "    save_tokenizer,\n",
    ")\n",
    "\n",
    "# Initial text\n",
    "text = \"Hello, world! This is a sample text for testing.\"\n",
    "torture_text = \"Hello, world! café پاکستان 日本語 🤖 ∇²ψ\"  # Cursed unicode string as a stress test"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "050c38e3",
   "metadata": {},
   "source": [
    "# Implementing a naive tokenizer"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
   "id": "c80d45c8",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:31.191303Z",
     "iopub.status.busy": "2026-08-23T00:04:31.191124Z",
     "iopub.status.idle": "2026-08-23T00:04:31.195638Z",
     "shell.execute_reply": "2026-08-23T00:04:31.194956Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Encoded: [4, 8, 13, 13, 16, 0, 21, 16, 18, 13, 7, 1, 0, 5, 11, 12, 19, 0, 12, 19, 0, 6, 0, 20, 8, 19, 20, 3]\n",
      "Decoded: Hello world! This is a test.\n"
     ]
    }
   ],
   "source": [
    "chars = sorted(set(text))\n",
    "\n",
    "char_to_index = {char: index for index, char in enumerate(chars)}\n",
    "index_to_char = {index: char for index, char in enumerate(chars)}\n",
    "\n",
    "\n",
    "def naive_encode(text):\n",
    "    # This function encodes the input text into a list of tokens.\n",
    "    return [char_to_index[char] for char in text]\n",
    "\n",
    "\n",
    "def naive_decode(tokens):\n",
    "    # This function decodes the list of tokens back into the original text.\n",
    "    return \"\".join(index_to_char[token] for token in tokens)\n",
    "\n",
    "\n",
    "encoded_text = naive_encode(\"Hello world! This is a test.\")\n",
    "print(\"Encoded:\", encoded_text)\n",
    "decoded_text = naive_decode(encoded_text)\n",
    "print(\"Decoded:\", decoded_text)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "ad064106",
   "metadata": {},
   "source": [
    "# Byte-level tokenizer\n"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
   "id": "98c6fb61",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:31.198093Z",
     "iopub.status.busy": "2026-08-23T00:04:31.197949Z",
     "iopub.status.idle": "2026-08-23T00:04:31.200411Z",
     "shell.execute_reply": "2026-08-23T00:04:31.199954Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Encoded torture text: [72, 101, 108, 108, 111, 44, 32, 119, 111, 114, 108, 100, 33, 32, 99, 97, 102, 195, 169, 32, 217, 190, 216, 167, 218, 169, 216, 179, 216, 170, 216, 167, 217, 134, 32, 230, 151, 165, 230, 156, 172, 232, 170, 158, 32, 240, 159, 164, 150, 32, 226, 136, 135, 194, 178, 207, 136]\n",
      "Decoded torture text: Hello, world! café پاکستان 日本語 🤖 ∇²ψ\n"
     ]
    }
   ],
   "source": [
    "# Unicode string → UTF-8 bytes → integer tokens\n",
    "\n",
    "encoded_torture_text = encode_unicode_string(torture_text)\n",
    "print(\"Encoded torture text:\", encoded_torture_text)\n",
    "decoded_torture_text = decode_unicode_tokens(encoded_torture_text)\n",
    "print(\"Decoded torture text:\", decoded_torture_text)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "ed28fbbd",
   "metadata": {},
   "source": [
    "# BPE (byte-pair encoding)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 4,
   "id": "2adedab3",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:31.201899Z",
     "iopub.status.busy": "2026-08-23T00:04:31.201786Z",
     "iopub.status.idle": "2026-08-23T00:04:31.204514Z",
     "shell.execute_reply": "2026-08-23T00:04:31.204064Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Final tokens: [310]\n"
     ]
    }
   ],
   "source": [
    "target_vocab_size = 512\n",
    "\n",
    "tokens = encode_unicode_string(torture_text)\n",
    "\n",
    "\n",
    "tokens, vocab, merges = perform_merges(tokens, target_vocab_size)\n",
    "print(\"Final tokens:\", tokens)"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 5,
   "id": "c4d8a1e6",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:31.205882Z",
     "iopub.status.busy": "2026-08-23T00:04:31.205783Z",
     "iopub.status.idle": "2026-08-23T00:04:33.966554Z",
     "shell.execute_reply": "2026-08-23T00:04:33.966095Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Encoded corpus into 47,586 tokens\n",
      "Saved tokenizer to data/tiny_shakespeare_bpe.json\n"
     ]
    }
   ],
   "source": [
    "# Tiny Shakespeare corpus\n",
    "corpus = (repo_root / \"data\" / \"tiny_shakespeare.txt\").read_text(encoding=\"utf-8\")\n",
    "\n",
    "tokens = encode_unicode_string(\n",
    "    corpus[:100_000]\n",
    ")  # Limit to first 100,000 characters for demonstration\n",
    "\n",
    "fin_tokens, vocab, merges = perform_merges(tokens, target_vocab_size)\n",
    "save_tokenizer(repo_root / \"data\" / \"tiny_shakespeare_bpe.json\", vocab, merges)\n",
    "\n",
    "print(f\"Encoded corpus into {len(fin_tokens):,} tokens\")\n",
    "print(\"Saved tokenizer to data/tiny_shakespeare_bpe.json\")"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
   "id": "267cbc2e",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:33.967721Z",
     "iopub.status.busy": "2026-08-23T00:04:33.967644Z",
     "iopub.status.idle": "2026-08-23T00:04:33.969895Z",
     "shell.execute_reply": "2026-08-23T00:04:33.969541Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Original: To be, or not to be 🤖\n",
      "Encoded: [84, 274, 317, 261, 470, 331, 294, 353, 240, 159, 164, 150]\n",
      "Decoded: To be, or not to be 🤖\n"
     ]
    }
   ],
   "source": [
    "test = \"To be, or not to be 🤖\"\n",
    "\n",
    "encoded = bpe_encode(test, vocab, merges)\n",
    "decoded = bpe_decode(encoded, vocab)\n",
    "print(\"Original:\", test)\n",
    "print(\"Encoded:\", encoded)\n",
    "print(\"Decoded:\", decoded)\n",
    "\n",
    "assert test == decoded, \"Decoded text does not match the original!\""
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 7,
   "id": "a16ab95b",
   "metadata": {
    "execution": {
     "iopub.execute_input": "2026-08-23T00:04:33.970993Z",
     "iopub.status.busy": "2026-08-23T00:04:33.970929Z",
     "iopub.status.idle": "2026-08-23T00:04:33.974289Z",
     "shell.execute_reply": "2026-08-23T00:04:33.974016Z"
    }
   },
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
      "Original: hello world\n",
      "Encoded: [323, 276, 274, 414, 108, 100]\n",
      "Decoded: hello world\n",
      "Original: To be, or not to be\n",
      "Encoded: [84, 274, 317, 261, 470, 331, 294, 317]\n",
      "Decoded: To be, or not to be\n",
      "Original: aaaaaaaaaaaa\n",
      "Encoded: [97, 97, 97, 97, 97, 97, 97, 97, 97, 97, 97, 97]\n",
      "Decoded: aaaaaaaaaaaa\n",
      "Original: hello\n",
      "world\n",
      "Encoded: [323, 276, 111, 10, 414, 108, 100]\n",
      "Decoded: hello\n",
      "world\n",
      "Original: 🐸\n",
      "Encoded: [240, 159, 144, 184]\n",
      "Decoded: 🐸\n",
      "Original: پاکستان\n",
      "Encoded: [217, 190, 216, 167, 218, 169, 216, 179, 216, 170, 216, 167, 217, 134]\n",
      "Decoded: پاکستان\n",
      "Original: hello 🐸 پاکستان\n",
      "Encoded: [323, 276, 274, 240, 159, 144, 184, 32, 217, 190, 216, 167, 218, 169, 216, 179, 216, 170, 216, 167, 217, 134]\n",
      "Decoded: hello 🐸 پاکستان\n",
      "Original: \n",
      "Encoded: []\n",
      "Decoded: \n"
     ]
    }
   ],
   "source": [
    "tests = [\n",
    "    \"hello world\",\n",
    "    \"To be, or not to be\",\n",
    "    \"aaaaaaaaaaaa\",\n",
    "    \"hello\\nworld\",\n",
    "    \"🐸\",\n",
    "    \"پاکستان\",\n",
    "    \"hello 🐸 پاکستان\",\n",
    "    \"\",\n",
    "]\n",
    "\n",
    "for test in tests:\n",
    "    encoded = bpe_encode(test, vocab, merges)\n",
    "    decoded = bpe_decode(encoded, vocab)\n",
    "    print(\"Original:\", test)\n",
    "    print(\"Encoded:\", encoded)\n",
    "    print(\"Decoded:\", decoded)\n",
    "    assert test == decoded, \"Decoded text does not match the original!\""
   ]
  }
 ],
 "metadata": {
  "kernelspec": {
   "display_name": "lmlab (3.12.13)",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "codemirror_mode": {
    "name": "ipython",
    "version": 3
   },
   "file_extension": ".py",
   "mimetype": "text/x-python",
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
   "version": "3.12.13"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
