In [1]:
from IPython.display import Image

Image(filename='UC-NEGRO-01.jpg', width=200)
Out[1]:
No description has been provided for this image

PROCESAMIENTO DE LENGUAJE NATURAL¶

  • Versiones de librerías, python 3.8.10

  • numpy 1.23.5

  • torch 1.10.0

In [2]:
import math
import torch
import torch.nn as nn
from torch.nn import functional as F

from utils import CfgNode as CN

Declaramos el módulo de atención causal¶

In [3]:
class NewGELU(nn.Module):
    def forward(self, x):
        return 0.5 * x * (1.0 + torch.tanh(math.sqrt(2.0 / math.pi) * (x + 0.044715 * torch.pow(x, 3.0))))

class CausalSelfAttention(nn.Module):

    def __init__(self, config):
        super().__init__()
        assert config.n_embd % config.n_head == 0
        # key, query, value projections for all heads, but in a batch
        self.c_attn = nn.Linear(config.n_embd, 3 * config.n_embd)
        # output projection
        self.c_proj = nn.Linear(config.n_embd, config.n_embd)
        # regularization
        self.attn_dropout = nn.Dropout(config.attn_pdrop)
        self.resid_dropout = nn.Dropout(config.resid_pdrop)
        # causal mask to ensure that attention is only applied to the left in the input sequence
        self.register_buffer("bias", torch.tril(torch.ones(config.block_size, config.block_size))
                                     .view(1, 1, config.block_size, config.block_size))
        self.n_head = config.n_head
        self.n_embd = config.n_embd

    def forward(self, x):
        B, T, C = x.size() # batch size, sequence length, embedding dimensionality (n_embd)

        # calculate query, key, values for all heads in batch and move head forward to be the batch dim
        q, k ,v  = self.c_attn(x).split(self.n_embd, dim=2)
        k = k.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)
        q = q.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)
        v = v.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)

        # causal self-attention; Self-attend: (B, nh, T, hs) x (B, nh, hs, T) -> (B, nh, T, T)
        att = (q @ k.transpose(-2, -1)) * (1.0 / math.sqrt(k.size(-1)))
        att = att.masked_fill(self.bias[:,:,:T,:T] == 0, float('-inf'))
        att = F.softmax(att, dim=-1)
        att = self.attn_dropout(att)
        y = att @ v # (B, nh, T, T) x (B, nh, T, hs) -> (B, nh, T, hs)
        y = y.transpose(1, 2).contiguous().view(B, T, C) # re-assemble all head outputs side by side

        # output projection
        y = self.resid_dropout(self.c_proj(y))
        return y

Aquí se declara un bloque transformer decoder¶

In [4]:
class Block(nn.Module):

    def __init__(self, config):
        super().__init__()
        self.ln_1 = nn.LayerNorm(config.n_embd)
        self.attn = CausalSelfAttention(config)
        self.ln_2 = nn.LayerNorm(config.n_embd)
        self.mlp = nn.ModuleDict(dict(
            c_fc    = nn.Linear(config.n_embd, 4 * config.n_embd),
            c_proj  = nn.Linear(4 * config.n_embd, config.n_embd),
            act     = NewGELU(),
            dropout = nn.Dropout(config.resid_pdrop),
        ))
        m = self.mlp
        self.mlpf = lambda x: m.dropout(m.c_proj(m.act(m.c_fc(x)))) # MLP forward

    def forward(self, x):
        x = x + self.attn(self.ln_1(x))
        x = x + self.mlpf(self.ln_2(x))
        return x

La clase GPT incluye la definición de varios tipos de modelos según los parámetros arquitectónicos¶

In [14]:
class GPT(nn.Module):
    """ GPT Language Model """

    @staticmethod
    def get_default_config():
        C = CN()
        # either model_type or (n_layer, n_head, n_embd) must be given in the config
        C.model_type = 'gpt'
        C.n_layer = None
        C.n_head = None
        C.n_embd =  None
        # these options must be filled in externally
        C.vocab_size = None
        C.block_size = None
        # dropout hyperparameters
        C.embd_pdrop = 0.1
        C.resid_pdrop = 0.1
        C.attn_pdrop = 0.1
        return C
    
    
    def __init__(self, config):
        super().__init__()
        assert config.vocab_size is not None
        assert config.block_size is not None
        self.block_size = config.block_size

        type_given = config.model_type is not None
        params_given = all([config.n_layer is not None, config.n_head is not None, config.n_embd is not None])
        assert type_given ^ params_given # exactly one of these (XOR)
        if type_given:
            # translate from model_type to detailed configuration
            config.merge_from_dict({
                # names follow the huggingface naming conventions
                # GPT-1
                'openai-gpt':   dict(n_layer=12, n_head=12, n_embd=768),  # 117M params
                # GPT-2 configs
                'gpt2':         dict(n_layer=12, n_head=12, n_embd=768),  # 124M params
                'gpt2-medium':  dict(n_layer=24, n_head=16, n_embd=1024), # 350M params
                'gpt2-large':   dict(n_layer=36, n_head=20, n_embd=1280), # 774M params
                'gpt2-xl':      dict(n_layer=48, n_head=25, n_embd=1600), # 1558M params
                # Gophers
                'gopher-44m':   dict(n_layer=8, n_head=16, n_embd=512),
                # (there are a number more...)
                # I made these tiny models up
                'gpt-mini':     dict(n_layer=6, n_head=6, n_embd=192),
                'gpt-micro':    dict(n_layer=4, n_head=4, n_embd=128),
                'gpt-nano':     dict(n_layer=3, n_head=3, n_embd=48),
            }[config.model_type])

        self.transformer = nn.ModuleDict(dict(
            wte = nn.Embedding(config.vocab_size, config.n_embd),
            wpe = nn.Embedding(config.block_size, config.n_embd),
            drop = nn.Dropout(config.embd_pdrop),
            h = nn.ModuleList([Block(config) for _ in range(config.n_layer)]),
            ln_f = nn.LayerNorm(config.n_embd),
        ))
        self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)

        # init all weights, and apply a special scaled init to the residual projections, per GPT-2 paper
        self.apply(self._init_weights)
        for pn, p in self.named_parameters():
            if pn.endswith('c_proj.weight'):
                torch.nn.init.normal_(p, mean=0.0, std=0.02/math.sqrt(2 * config.n_layer))

        # report number of parameters (note we don't count the decoder parameters in lm_head)
        n_params = sum(p.numel() for p in self.transformer.parameters())
        print("number of parameters: %.2fM" % (n_params/1e6,))
    
    def _init_weights(self, module):
        if isinstance(module, nn.Linear):
            torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
            if module.bias is not None:
                torch.nn.init.zeros_(module.bias)
        elif isinstance(module, nn.Embedding):
            torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
        elif isinstance(module, nn.LayerNorm):
            torch.nn.init.zeros_(module.bias)
            torch.nn.init.ones_(module.weight)
    
    @classmethod
    def from_pretrained(cls, model_type):
        """
        Initialize a pretrained GPT model by copying over the weights
        from a huggingface/transformers checkpoint.
        """
        assert model_type in {'gpt2', 'gpt2-medium', 'gpt2-large', 'gpt2-xl'}
        from transformers import GPT2LMHeadModel

        # create a from-scratch initialized minGPT model
        config = cls.get_default_config()
        config.model_type = model_type
        config.vocab_size = 50257 # openai's model vocabulary
        config.block_size = 1024  # openai's model block_size
        model = GPT(config)
        sd = model.state_dict()

        # init a huggingface/transformers model
        model_hf = GPT2LMHeadModel.from_pretrained(model_type)
        sd_hf = model_hf.state_dict()

        # copy while ensuring all of the parameters are aligned and match in names and shapes
        keys = [k for k in sd_hf if not k.endswith('attn.masked_bias')] # ignore these
        transposed = ['attn.c_attn.weight', 'attn.c_proj.weight', 'mlp.c_fc.weight', 'mlp.c_proj.weight']
        # basically the openai checkpoints use a "Conv1D" module, but we only want to use a vanilla nn.Linear.
        # this means that we have to transpose these weights when we import them
        assert len(keys) == len(sd)
        for k in keys:
            if any(k.endswith(w) for w in transposed):
                # special treatment for the Conv1D weights we need to transpose
                assert sd_hf[k].shape[::-1] == sd[k].shape
                with torch.no_grad():
                    sd[k].copy_(sd_hf[k].t())
            else:
                # vanilla copy over the other parameters
                assert sd_hf[k].shape == sd[k].shape
                with torch.no_grad():
                    sd[k].copy_(sd_hf[k])

        return model
    
    def configure_optimizers(self, train_config):
        """
        This long function is unfortunately doing something very simple and is being very defensive:
        We are separating out all parameters of the model into two buckets: those that will experience
        weight decay for regularization and those that won't (biases, and layernorm/embedding weights).
        We are then returning the PyTorch optimizer object.
        """

        # separate out all parameters to those that will and won't experience regularizing weight decay
        decay = set()
        no_decay = set()
        whitelist_weight_modules = (torch.nn.Linear, )
        blacklist_weight_modules = (torch.nn.LayerNorm, torch.nn.Embedding)
        for mn, m in self.named_modules():
            for pn, p in m.named_parameters():
                fpn = '%s.%s' % (mn, pn) if mn else pn # full param name
                # random note: because named_modules and named_parameters are recursive
                # we will see the same tensors p many many times. but doing it this way
                # allows us to know which parent module any tensor p belongs to...
                if pn.endswith('bias'):
                    # all biases will not be decayed
                    no_decay.add(fpn)
                elif pn.endswith('weight') and isinstance(m, whitelist_weight_modules):
                    # weights of whitelist modules will be weight decayed
                    decay.add(fpn)
                elif pn.endswith('weight') and isinstance(m, blacklist_weight_modules):
                    # weights of blacklist modules will NOT be weight decayed
                    no_decay.add(fpn)

        # validate that we considered every parameter
        param_dict = {pn: p for pn, p in self.named_parameters()}
        inter_params = decay & no_decay
        union_params = decay | no_decay
        assert len(inter_params) == 0, "parameters %s made it into both decay/no_decay sets!" % (str(inter_params), )
        assert len(param_dict.keys() - union_params) == 0, "parameters %s were not separated into either decay/no_decay set!" \
                                                    % (str(param_dict.keys() - union_params), )

        # create the pytorch optimizer object
        optim_groups = [
            {"params": [param_dict[pn] for pn in sorted(list(decay))], "weight_decay": train_config.weight_decay},
            {"params": [param_dict[pn] for pn in sorted(list(no_decay))], "weight_decay": 0.0},
        ]
        optimizer = torch.optim.AdamW(optim_groups, lr=train_config.learning_rate, betas=train_config.betas)
        return optimizer
    
    
    def forward(self, idx, targets=None):
        device = idx.device
        b, t = idx.size()
        assert t <= self.block_size, f"Cannot forward sequence of length {t}, block size is only {self.block_size}"
        pos = torch.arange(0, t, dtype=torch.long, device=device).unsqueeze(0) # shape (1, t)

        # forward the GPT model itself
        tok_emb = self.transformer.wte(idx) # token embeddings of shape (b, t, n_embd)
        pos_emb = self.transformer.wpe(pos) # position embeddings of shape (1, t, n_embd)
        x = self.transformer.drop(tok_emb + pos_emb)
        for block in self.transformer.h:
            x = block(x)
        x = self.transformer.ln_f(x)
        logits = self.lm_head(x)

        # if we are given some desired targets also calculate the loss
        loss = None
        if targets is not None:
            loss = F.cross_entropy(logits.view(-1, logits.size(-1)), targets.view(-1), ignore_index=-1)

        return logits, loss
    
    
    @torch.no_grad()
    def generate(self, idx, max_new_tokens, temperature=1.0, do_sample=False, top_k=None):
        """
        Take a conditioning sequence of indices idx (LongTensor of shape (b,t)) and complete
        the sequence max_new_tokens times, feeding the predictions back into the model each time.
        Most likely you'll want to make sure to be in model.eval() mode of operation for this.
        """
        for _ in range(max_new_tokens):
            # if the sequence context is growing too long we must crop it at block_size
            idx_cond = idx if idx.size(1) <= self.block_size else idx[:, -self.block_size:]
            # forward the model to get the logits for the index in the sequence
            logits, _ = self(idx_cond)
            # pluck the logits at the final step and scale by desired temperature
            logits = logits[:, -1, :] / temperature
            # optionally crop the logits to only the top k options
            if top_k is not None:
                v, _ = torch.topk(logits, top_k)
                logits[logits < v[:, [-1]]] = -float('Inf')
            # apply softmax to convert logits to (normalized) probabilities
            probs = F.softmax(logits, dim=-1)
            # either sample from the distribution or take the most likely element
            if do_sample:
                idx_next = torch.multinomial(probs, num_samples=1)
            else:
                _, idx_next = torch.topk(probs, k=1, dim=-1)
            # append sampled index to the running sequence and continue
            idx = torch.cat((idx, idx_next), dim=1)

        return idx

El trainer implementa el forward sobre los datos de train.¶

In [6]:
import time
from collections import defaultdict
from torch.utils.data.dataloader import DataLoader

class Trainer:

    @staticmethod
    def get_default_config():
        C = CN()
        # device to train on
        #C.device = 'auto'
        C.device = 'cpu'
        # dataloder parameters
        C.num_workers = 4
        # optimizer parameters
        C.max_iters = None
        C.batch_size = 64
        C.learning_rate = 3e-4
        C.betas = (0.9, 0.95)
        C.weight_decay = 0.1
        C.grad_norm_clip = 1.0
        return C

    def __init__(self, config, model, train_dataset):
        self.config = config
        self.model = model
        self.optimizer = None
        self.train_dataset = train_dataset
        self.callbacks = defaultdict(list)

        # determine the device we'll train on
        if config.device == 'auto':
            self.device = 'cuda' if torch.cuda.is_available() else 'cpu'
        else:
            self.device = config.device
        self.model = self.model.to(self.device)
        print("running on device", self.device)

        # variables that will be assigned to trainer class later for logging and etc
        self.iter_num = 0
        self.iter_time = 0.0
        self.iter_dt = 0.0

    def add_callback(self, onevent: str, callback):
        self.callbacks[onevent].append(callback)

    def set_callback(self, onevent: str, callback):
        self.callbacks[onevent] = [callback]

    def trigger_callbacks(self, onevent: str):
        for callback in self.callbacks.get(onevent, []):
            callback(self)

    def run(self):
        model, config = self.model, self.config

        # setup the optimizer
        self.optimizer = model.configure_optimizers(config)

        # setup the dataloader
        train_loader = DataLoader(
            self.train_dataset,
            sampler=torch.utils.data.RandomSampler(self.train_dataset, replacement=True, num_samples=int(1e10)),
            shuffle=False,
            pin_memory=True,
            batch_size=config.batch_size,
            num_workers=config.num_workers,
        )

        model.train()
        self.iter_num = 0
        self.iter_time = time.time()
        data_iter = iter(train_loader)
        while True:

            # fetch the next batch (x, y) and re-init iterator if needed
            try:
                batch = next(data_iter)
            except StopIteration:
                data_iter = iter(train_loader)
                batch = next(data_iter)
            batch = [t.to(self.device) for t in batch]
            x, y = batch

            # forward the model
            logits, self.loss = model(x, y)

            # backprop and update the parameters
            model.zero_grad(set_to_none=True)
            self.loss.backward()
            torch.nn.utils.clip_grad_norm_(model.parameters(), config.grad_norm_clip)
            self.optimizer.step()

            self.trigger_callbacks('on_batch_end')
            self.iter_num += 1
            tnow = time.time()
            self.iter_dt = tnow - self.iter_time
            self.iter_time = tnow

            # termination conditions
            if config.max_iters is not None and self.iter_num >= config.max_iters:
                break

En este ejemplo, la tarea consiste en ordenar una secuencia de enteros. Esto tiene particularidades ya que usaremos una máscara 'I' para indicar al transformer que estamos ingresando la entrada, estrategia denominada prefix tuning.¶

In [7]:
from torch.utils.data import Dataset
from torch.utils.data.dataloader import DataLoader

import pickle

class SortDataset(Dataset):
    """ 
    Dataset for the Sort problem. E.g. for problem length 6:
    Input: 0 0 2 1 0 1 -> Output: 0 0 0 1 1 2
    Which will feed into the transformer concatenated as:
    input:  0 0 2 1 0 1 0 0 0 1 1
    output: I I I I I 0 0 0 1 1 2
    where I is "ignore", as the transformer is reading the input sequence
    """

    def __init__(self, split, length=6, num_digits=3):
        assert split in {'train', 'test'}
        self.split = split
        self.length = length
        self.num_digits = num_digits
    
    def __len__(self):
        return 10000 # ...
    
    def get_vocab_size(self):
        return self.num_digits
    
    def get_block_size(self):
        # the length of the sequence that will feed into transformer, 
        # containing concatenated input and the output, but -1 because
        # the transformer starts making predictions at the last input element
        return self.length * 2 - 1

    def __getitem__(self, idx):
        
        # use rejection sampling to generate an input example from the desired split
        while True:
            # generate some random integers
            inp = torch.randint(self.num_digits, size=(self.length,), dtype=torch.long)
            # half of the time let's try to boost the number of examples that 
            # have a large number of repeats, as this is what the model seems to struggle
            # with later in training, and they are kind of rate
            if torch.rand(1).item() < 0.5:
                if inp.unique().nelement() > self.length // 2:
                    # too many unqiue digits, re-sample
                    continue
            # figure out if this generated example is train or test based on its hash
            h = hash(pickle.dumps(inp.tolist()))
            inp_split = 'test' if h % 4 == 0 else 'train' # designate 25% of examples as test
            if inp_split == self.split:
                break # ok
        
        # solve the task: i.e. sort
        sol = torch.sort(inp)[0]

        # concatenate the problem specification and the solution
        cat = torch.cat((inp, sol), dim=0)

        # the inputs to the transformer will be the offset sequence
        x = cat[:-1].clone()
        y = cat[1:].clone()
        # we only want to predict at output locations, mask out the loss at the input locations
        y[:self.length-1] = -1
        return x, y
In [8]:
train_dataset = SortDataset('train')
test_dataset = SortDataset('test')
x, y = train_dataset[0]
for a, b in zip(x,y):
    print(int(a),int(b))
2 -1
1 -1
0 -1
0 -1
2 -1
2 0
0 0
0 1
1 2
2 2
2 2

Este es un modelo pequeño, ya que lo vamos a entrenar en cpu para simplificar el tema de acceso a gpus¶

In [9]:
model_config = GPT.get_default_config()
model_config.model_type = 'gpt-nano'
model_config.vocab_size = train_dataset.get_vocab_size()
model_config.block_size = train_dataset.get_block_size()
model = GPT(model_config)
number of parameters: 0.09M
In [10]:
CUDA_LAUNCH_BLOCKING=1
train_config = Trainer.get_default_config()
train_config.learning_rate = 5e-4 
train_config.max_iters = 10000
train_config.num_workers = 0
trainer = Trainer(train_config, model, train_dataset)
running on device cpu
In [11]:
def batch_end_callback(trainer):
    if trainer.iter_num % 100 == 0:
        print(f"iter_dt {trainer.iter_dt * 1000:.2f}ms; iter {trainer.iter_num}: train loss {trainer.loss.item():.5f}")
trainer.set_callback('on_batch_end', batch_end_callback)

trainer.run()
iter_dt 0.00ms; iter 0: train loss 1.14617
iter_dt 16.91ms; iter 100: train loss 0.16010
iter_dt 22.04ms; iter 200: train loss 0.06826
iter_dt 15.27ms; iter 300: train loss 0.06172
iter_dt 22.94ms; iter 400: train loss 0.04327
iter_dt 17.05ms; iter 500: train loss 0.02389
iter_dt 15.96ms; iter 600: train loss 0.02782
iter_dt 17.86ms; iter 700: train loss 0.00776
iter_dt 17.04ms; iter 800: train loss 0.01967
iter_dt 26.84ms; iter 900: train loss 0.01876
iter_dt 16.68ms; iter 1000: train loss 0.02067
iter_dt 15.81ms; iter 1100: train loss 0.01211
iter_dt 16.99ms; iter 1200: train loss 0.01564
iter_dt 16.60ms; iter 1300: train loss 0.00063
iter_dt 16.71ms; iter 1400: train loss 0.00075
iter_dt 16.56ms; iter 1500: train loss 0.00899
iter_dt 17.18ms; iter 1600: train loss 0.01194
iter_dt 17.26ms; iter 1700: train loss 0.01115
iter_dt 16.26ms; iter 1800: train loss 0.00106
iter_dt 15.51ms; iter 1900: train loss 0.01863
iter_dt 18.32ms; iter 2000: train loss 0.00370
iter_dt 16.28ms; iter 2100: train loss 0.00263
iter_dt 16.87ms; iter 2200: train loss 0.00784
iter_dt 17.85ms; iter 2300: train loss 0.00167
iter_dt 16.53ms; iter 2400: train loss 0.00545
iter_dt 24.46ms; iter 2500: train loss 0.00599
iter_dt 17.15ms; iter 2600: train loss 0.00060
iter_dt 15.42ms; iter 2700: train loss 0.00168
iter_dt 16.85ms; iter 2800: train loss 0.00045
iter_dt 26.57ms; iter 2900: train loss 0.02349
iter_dt 16.13ms; iter 3000: train loss 0.00041
iter_dt 16.44ms; iter 3100: train loss 0.00121
iter_dt 16.10ms; iter 3200: train loss 0.02541
iter_dt 21.17ms; iter 3300: train loss 0.02491
iter_dt 16.09ms; iter 3400: train loss 0.00166
iter_dt 23.06ms; iter 3500: train loss 0.01533
iter_dt 23.91ms; iter 3600: train loss 0.00018
iter_dt 25.28ms; iter 3700: train loss 0.00279
iter_dt 17.21ms; iter 3800: train loss 0.00076
iter_dt 15.68ms; iter 3900: train loss 0.00276
iter_dt 25.46ms; iter 4000: train loss 0.00343
iter_dt 14.45ms; iter 4100: train loss 0.00077
iter_dt 17.21ms; iter 4200: train loss 0.00149
iter_dt 16.79ms; iter 4300: train loss 0.00072
iter_dt 15.74ms; iter 4400: train loss 0.01602
iter_dt 17.56ms; iter 4500: train loss 0.00027
iter_dt 16.34ms; iter 4600: train loss 0.00031
iter_dt 16.31ms; iter 4700: train loss 0.00040
iter_dt 19.95ms; iter 4800: train loss 0.00219
iter_dt 19.18ms; iter 4900: train loss 0.04403
iter_dt 28.14ms; iter 5000: train loss 0.00515
iter_dt 30.79ms; iter 5100: train loss 0.00121
iter_dt 20.64ms; iter 5200: train loss 0.00155
iter_dt 24.16ms; iter 5300: train loss 0.00013
iter_dt 16.49ms; iter 5400: train loss 0.00057
iter_dt 17.62ms; iter 5500: train loss 0.00095
iter_dt 16.52ms; iter 5600: train loss 0.00009
iter_dt 21.65ms; iter 5700: train loss 0.01718
iter_dt 19.77ms; iter 5800: train loss 0.00049
iter_dt 23.02ms; iter 5900: train loss 0.00028
iter_dt 16.60ms; iter 6000: train loss 0.00568
iter_dt 16.45ms; iter 6100: train loss 0.00390
iter_dt 29.99ms; iter 6200: train loss 0.02280
iter_dt 16.08ms; iter 6300: train loss 0.00080
iter_dt 16.39ms; iter 6400: train loss 0.00013
iter_dt 17.62ms; iter 6500: train loss 0.01391
iter_dt 20.30ms; iter 6600: train loss 0.00114
iter_dt 21.15ms; iter 6700: train loss 0.00005
iter_dt 20.87ms; iter 6800: train loss 0.00590
iter_dt 17.20ms; iter 6900: train loss 0.00297
iter_dt 30.51ms; iter 7000: train loss 0.00230
iter_dt 16.80ms; iter 7100: train loss 0.00382
iter_dt 16.53ms; iter 7200: train loss 0.00147
iter_dt 27.03ms; iter 7300: train loss 0.00871
iter_dt 20.12ms; iter 7400: train loss 0.01805
iter_dt 16.47ms; iter 7500: train loss 0.01507
iter_dt 15.83ms; iter 7600: train loss 0.00013
iter_dt 15.69ms; iter 7700: train loss 0.00010
iter_dt 15.69ms; iter 7800: train loss 0.00023
iter_dt 15.49ms; iter 7900: train loss 0.00037
iter_dt 27.93ms; iter 8000: train loss 0.00015
iter_dt 15.68ms; iter 8100: train loss 0.00082
iter_dt 16.84ms; iter 8200: train loss 0.00012
iter_dt 17.36ms; iter 8300: train loss 0.00007
iter_dt 16.18ms; iter 8400: train loss 0.00006
iter_dt 16.50ms; iter 8500: train loss 0.00010
iter_dt 17.78ms; iter 8600: train loss 0.00021
iter_dt 17.23ms; iter 8700: train loss 0.00011
iter_dt 17.15ms; iter 8800: train loss 0.00048
iter_dt 16.67ms; iter 8900: train loss 0.00545
iter_dt 16.85ms; iter 9000: train loss 0.00007
iter_dt 16.48ms; iter 9100: train loss 0.00003
iter_dt 16.49ms; iter 9200: train loss 0.00015
iter_dt 15.75ms; iter 9300: train loss 0.00626
iter_dt 20.84ms; iter 9400: train loss 0.00088
iter_dt 17.90ms; iter 9500: train loss 0.00047
iter_dt 16.40ms; iter 9600: train loss 0.00042
iter_dt 31.59ms; iter 9700: train loss 0.00012
iter_dt 18.27ms; iter 9800: train loss 0.00790
iter_dt 26.44ms; iter 9900: train loss 0.00005

En eval split comparamos la secuencia entregada por gpt con la secuencia ordenada y medimos los errores.¶

In [13]:
def eval_split(trainer, split, max_batches):
    dataset = {'train':train_dataset, 'test':test_dataset}[split]
    n = train_dataset.length 
    print(n)
    results = []
    mistakes_printed_already = 0
    loader = DataLoader(dataset, batch_size=100, num_workers=0, drop_last=False)
    for b, (x, y) in enumerate(loader):
        x = x.to(trainer.device)
        y = y.to(trainer.device)
        # isolate the input pattern alone
        inp = x[:, :n]
        sol = y[:, -n:]
        # let the model sample the rest of the sequence
        cat = model.generate(inp, n, do_sample=False) # using greedy argmax, not sampling
        sol_candidate = cat[:, n:] # isolate the filled in sequence
        # compare the predicted sequence to the true sequence
        correct = (sol == sol_candidate).all(1).cpu() 
        for i in range(x.size(0)):
            results.append(int(correct[i]))
            if not correct[i] and mistakes_printed_already < 3: # only print up to 5 mistakes to get a sense
                mistakes_printed_already += 1
                print("GPT claims that %s sorted is %s but gt is %s" % (inp[i].tolist(), sol_candidate[i].tolist(), sol[i].tolist()))
        if max_batches is not None and b+1 >= max_batches:
            break
    rt = torch.tensor(results, dtype=torch.float)
    print("%s final score: %d/%d = %.2f%% correct" % (split, rt.sum(), len(results), 100*rt.mean()))
    return rt.sum()

# run a lot of examples from both train and test through the model and verify the output correctness
with torch.no_grad():
    train_score = eval_split(trainer, 'train', max_batches=50)
    test_score  = eval_split(trainer, 'test',  max_batches=50)
6
GPT claims that [0, 1, 0, 1, 2, 1] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2]
GPT claims that [0, 2, 1, 1, 0, 1] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2]
GPT claims that [0, 1, 1, 1, 0, 2] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2]
train final score: 4977/5000 = 99.54% correct
6
GPT claims that [0, 1, 1, 2, 0, 0] sorted is [0, 0, 1, 1, 1, 2] but gt is [0, 0, 0, 1, 1, 2]
GPT claims that [1, 2, 2, 1, 2, 2] sorted is [1, 2, 2, 2, 2, 2] but gt is [1, 1, 2, 2, 2, 2]
GPT claims that [2, 2, 2, 1, 2, 2] sorted is [2, 2, 2, 2, 2, 2] but gt is [1, 2, 2, 2, 2, 2]
test final score: 4987/5000 = 99.74% correct
In [13]:
n = train_dataset.length 
inp = torch.tensor([[0, 0, 2, 1, 0, 1]], dtype=torch.long).to(trainer.device)
assert inp[0].nelement() == n
with torch.no_grad():
    cat = model.generate(inp, n, do_sample=False)
sol = torch.sort(inp[0])[0]
sol_candidate = cat[:, n:]
print('input sequence  :', inp.tolist())
print('predicted sorted:', sol_candidate.tolist())
print('gt sort         :', sol.tolist())
print('matches         :', bool((sol == sol_candidate).all()))
input sequence  : [[0, 0, 2, 1, 0, 1]]
predicted sorted: [[0, 0, 0, 1, 1, 2]]
gt sort         : [0, 0, 0, 1, 1, 2]
matches         : True
In [ ]: