{ "cells": [ { "cell_type": "code", "execution_count": 2, "id": "193c3159", "metadata": {}, "outputs": [], "source": [ "with open(\"input.txt\", \"r\") as f:\n", " text = f.read()" ] }, { "cell_type": "code", "execution_count": 3, "id": "e557cb70", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "Length of text: 1115394 characters\n" ] } ], "source": [ "length = len(text)\n", "print(f\"Length of text: {length} characters\")" ] }, { "cell_type": "code", "execution_count": null, "id": "750587a9", "metadata": {}, "outputs": [], "source": [ "print(text[:500]) " ] }, { "cell_type": "code", "execution_count": 5, "id": "16490999", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "\n", " !$&',-.3:;?ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz\n", "Vocab size: 65\n" ] } ], "source": [ "char = sorted(list(set(text)))\n", "vocab_size = len(char)\n", "print(\"\".join(char))\n", "print(f\"Vocab size: {vocab_size}\")" ] }, { "cell_type": "code", "execution_count": 39, "id": "d9e6e17a", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "using mps device\n" ] } ], "source": [ "#use mps as i am using the mac with m4 \n", "device = \"mps\" if torch.backends.mps.is_available() else \"cpu\"\n", "print(f\"using {device} device\")" ] }, { "cell_type": "code", "execution_count": 6, "id": "082fd1ba", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "[46, 43, 50, 50, 53, 1, 61, 53, 56, 50, 42]\n", "hello world\n" ] } ], "source": [ "stoi = {ch:i for i,ch in enumerate(char)}\n", "itos = {i:ch for i,ch in enumerate(char)}\n", "encode = lambda s: [stoi[c] for c in s]\n", "decode = lambda l: \"\".join([itos[i] for i in l])\n", "print(encode(\"hello world\"))\n", "print(decode(encode(\"hello world\"))) # note this is one of the simplest possible tokenizers, it just maps each character to an integer. everyone has their own tokenizer like google use sentencepiece, openai use bpe, etc. we will build our own tokenizer in the next notebook." ] }, { "cell_type": "code", "execution_count": 7, "id": "7cce9365", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "torch.Size([1115394]) torch.int64\n", "tensor([18, 47, 56, 57, 58, 1, 15, 47, 58, 47, 64, 43, 52, 10, 0, 14, 43, 44,\n", " 53, 56, 43, 1, 61, 43, 1, 54, 56, 53, 41, 43, 43, 42, 1, 39, 52, 63,\n", " 1, 44, 59, 56, 58, 46, 43, 56, 6, 1, 46, 43, 39, 56, 1, 51, 43, 1,\n", " 57, 54, 43, 39, 49, 8, 0, 0, 13, 50, 50, 10, 0, 31, 54, 43, 39, 49,\n", " 6, 1, 57, 54, 43, 39, 49, 8, 0, 0, 18, 47, 56, 57, 58, 1, 15, 47,\n", " 58, 47, 64, 43, 52, 10, 0, 37, 53, 59])\n" ] } ], "source": [ "import torch\n", "data = torch.tensor(encode(text), dtype=torch.long)\n", "print(data.shape, data.dtype)\n", "print(data[:100])" ] }, { "cell_type": "code", "execution_count": 8, "id": "d59606cc", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "torch.Size([1003854]) torch.Size([111540])\n" ] } ], "source": [ "n = int(0.9*len(data))\n", "train_data = data[:n]\n", "val_data = data[n:]\n", "print(train_data.shape, val_data.shape)" ] }, { "cell_type": "code", "execution_count": 9, "id": "e2bd00e4", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "tensor([18, 47, 56, 57, 58, 1, 15, 47, 58])\n" ] } ], "source": [ "block_size = 8\n", "train_data[:block_size+1] # we will use the first 8 characters to predict\n", "print(train_data[:block_size+1])" ] }, { "cell_type": "code", "execution_count": 11, "id": "4ce6af03", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "when input is tensor([18]) the target: 47\n", "when input is tensor([18, 47]) the target: 56\n", "when input is tensor([18, 47, 56]) the target: 57\n", "when input is tensor([18, 47, 56, 57]) the target: 58\n", "when input is tensor([18, 47, 56, 57, 58]) the target: 1\n", "when input is tensor([18, 47, 56, 57, 58, 1]) the target: 15\n", "when input is tensor([18, 47, 56, 57, 58, 1, 15]) the target: 47\n", "when input is tensor([18, 47, 56, 57, 58, 1, 15, 47]) the target: 58\n" ] } ], "source": [ "x_train = train_data[:block_size]\n", "y_train = train_data[1:block_size+1]\n", "for t in range(block_size):\n", " context = x_train[:t+1]\n", " target = y_train[t]\n", " print(f\"when input is {context} the target: {target}\")" ] }, { "cell_type": "code", "execution_count": 38, "id": "85e56335", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "inputs:\n", "torch.Size([4, 8])\n", "tensor([[24, 43, 58, 5, 57, 1, 46, 43],\n", " [44, 53, 56, 1, 58, 46, 39, 58],\n", " [52, 58, 1, 58, 46, 39, 58, 1],\n", " [25, 17, 27, 10, 0, 21, 1, 54]], device='mps:0')\n", "\n", "targets:\n", "torch.Size([4, 8])\n", "tensor([[43, 58, 5, 57, 1, 46, 43, 39],\n", " [53, 56, 1, 58, 46, 39, 58, 1],\n", " [58, 1, 58, 46, 39, 58, 1, 46],\n", " [17, 27, 10, 0, 21, 1, 54, 39]], device='mps:0')\n" ] } ], "source": [ "torch.manual_seed(1337)\n", "batch_size = 4 # how many independent sequences will we process in parallel?\n", "block_size = 8 # what is the maximum context length for predictions?\n", "def get_batch(split):\n", " data = train_data if split == 'train' else val_data\n", " ix = torch.randint(len(data) - block_size, (batch_size,))\n", " x = torch.stack([data[i:i+block_size] for i in ix])\n", " y = torch.stack([data[i+1:i+block_size+1] for i in ix])\n", " x, y = x.to(device), y.to(device)\n", " return x, y\n", "xb, yb = get_batch('train')\n", "print(\"inputs:\")\n", "print(xb.shape)\n", "print(xb)\n", "print(\"\\ntargets:\")\n", "print(yb.shape)\n", "print(yb)" ] }, { "cell_type": "code", "execution_count": null, "id": "18810b27", "metadata": {}, "outputs": [], "source": [ "for b in range(batch_size):\n", " for t in range(block_size):\n", " context = xb[b, :t+1]\n", " target = yb[b, t]\n", " print(f\"when input is {context.tolist()} the target: {target.item()}\")" ] }, { "cell_type": "code", "execution_count": 22, "id": "77449b2f", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "tensor([[24, 43, 58, 5, 57, 1, 46, 43],\n", " [44, 53, 56, 1, 58, 46, 39, 58],\n", " [52, 58, 1, 58, 46, 39, 58, 1],\n", " [25, 17, 27, 10, 0, 21, 1, 54]])\n", "\n", "\n", "tensor([[43, 58, 5, 57, 1, 46, 43, 39],\n", " [53, 56, 1, 58, 46, 39, 58, 1],\n", " [58, 1, 58, 46, 39, 58, 1, 46],\n", " [17, 27, 10, 0, 21, 1, 54, 39]])\n" ] } ], "source": [ "print(xb)\n", "print(\"\\n\")\n", "print(yb)" ] }, { "cell_type": "code", "execution_count": null, "id": "66a1c195", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "logits shape: torch.Size([32, 65])\n", "loss: 4.878634929656982\n", "\n", "SKIcLT;AcE\n" ] } ], "source": [ "#Bigram language model\n", "import torch\n", "import torch.nn as nn\n", "import torch.nn.functional as F\n", "torch.manual_seed(1337)\n", "class BigramLanguageModel(torch.nn.Module):\n", " def __init__(self, vocab_size):\n", " super().__init__()\n", " self.token_embedding_table = torch.nn.Embedding(vocab_size, vocab_size)\n", " def forward(self, idx, targets=None):\n", " # idx and targets are both (B,T) tensor of integers\n", " logits = self.token_embedding_table(idx) # (B,T,C)\n", " if targets is None:\n", " loss = None\n", " else:\n", " B,T,C = logits.shape\n", " logits = logits.view(B*T, C)\n", " targets = targets.view(B*T)\n", " loss = F.cross_entropy(logits, targets)\n", " return logits, loss\n", " \n", " def generate(self, idx, max_new_tokens):\n", " # idx is (B,T) array of indices in the current context\n", " for _ in range(max_new_tokens):\n", " logits, loss = self(idx)\n", " logits = logits[:, -1, :] # becomes (B,C) , as we only want to provide the last character as the input to predict the next character\n", " probs = F.softmax(logits, dim=-1) # (B,C)\n", " idx_next = torch.multinomial(probs, num_samples=1) # (B,1)\n", " idx = torch.cat((idx, idx_next), dim=1) # (B,T+1)\n", " return idx\n", " \n", "model = BigramLanguageModel(vocab_size)\n", "model₹.to(device)\n", "logits, loss = model(xb, yb)\n", "print(\"logits shape:\", logits.shape)\n", "print(\"loss:\", loss.item())\n", "print(decode(model.generate(idx=torch.zeros((1,1), dtype=torch.long), max_new_tokens=10)[0].tolist()))" ] }, { "cell_type": "code", "execution_count": 35, "id": "ecd49fc4", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "step 9999: loss 2.4313366413116455\n" ] } ], "source": [ "optimizer = torch.optim.AdamW(model.parameters(), lr=1e-3)\n", "batch_size = 32\n", "for steps in range(10000):\n", " xb, yb = get_batch('train')\n", " logits, loss = model(xb, yb)\n", " optimizer.zero_grad(set_to_none=True)\n", " loss.backward()\n", " optimizer.step()\n", "print(f\"step {steps}: loss {loss.item()}\")\n", "\n" ] }, { "cell_type": "code", "execution_count": 36, "id": "8ce29e5a", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "\n", "Warstyo a \n" ] } ], "source": [ "print(decode(model.generate(idx=torch.zeros((1,1), dtype=torch.long), max_new_tokens=10)[0].tolist()))" ] }, { "cell_type": "code", "execution_count": null, "id": "b87bd156", "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ "using mps device\n" ] } ], "source": [] }, { "cell_type": "code", "execution_count": null, "id": "c9f2052b", "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": ".venv", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.14.2" } }, "nbformat": 4, "nbformat_minor": 5 }