In [1]:
from IPython.display import Image
Image(filename='UC-NEGRO-01.jpg', width=200)
Out[1]:
PROCESAMIENTO DE LENGUAJE NATURAL¶
Versiones de librerías, python 3.8.10
numpy 1.23.5
torch 1.10.0
In [2]:
import math
import torch
import torch.nn as nn
from torch.nn import functional as F
from utils import CfgNode as CN
Declaramos el módulo de atención causal¶
In [3]:
class NewGELU(nn.Module):
def forward(self, x):
return 0.5 * x * (1.0 + torch.tanh(math.sqrt(2.0 / math.pi) * (x + 0.044715 * torch.pow(x, 3.0))))
class CausalSelfAttention(nn.Module):
def __init__(self, config):
super().__init__()
assert config.n_embd % config.n_head == 0
# key, query, value projections for all heads, but in a batch
self.c_attn = nn.Linear(config.n_embd, 3 * config.n_embd)
# output projection
self.c_proj = nn.Linear(config.n_embd, config.n_embd)
# regularization
self.attn_dropout = nn.Dropout(config.attn_pdrop)
self.resid_dropout = nn.Dropout(config.resid_pdrop)
# causal mask to ensure that attention is only applied to the left in the input sequence
self.register_buffer("bias", torch.tril(torch.ones(config.block_size, config.block_size))
.view(1, 1, config.block_size, config.block_size))
self.n_head = config.n_head
self.n_embd = config.n_embd
def forward(self, x):
B, T, C = x.size() # batch size, sequence length, embedding dimensionality (n_embd)
# calculate query, key, values for all heads in batch and move head forward to be the batch dim
q, k ,v = self.c_attn(x).split(self.n_embd, dim=2)
k = k.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)
q = q.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)
v = v.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) # (B, nh, T, hs)
# causal self-attention; Self-attend: (B, nh, T, hs) x (B, nh, hs, T) -> (B, nh, T, T)
att = (q @ k.transpose(-2, -1)) * (1.0 / math.sqrt(k.size(-1)))
att = att.masked_fill(self.bias[:,:,:T,:T] == 0, float('-inf'))
att = F.softmax(att, dim=-1)
att = self.attn_dropout(att)
y = att @ v # (B, nh, T, T) x (B, nh, T, hs) -> (B, nh, T, hs)
y = y.transpose(1, 2).contiguous().view(B, T, C) # re-assemble all head outputs side by side
# output projection
y = self.resid_dropout(self.c_proj(y))
return y
Aquí se declara un bloque transformer decoder¶
In [4]:
class Block(nn.Module):
def __init__(self, config):
super().__init__()
self.ln_1 = nn.LayerNorm(config.n_embd)
self.attn = CausalSelfAttention(config)
self.ln_2 = nn.LayerNorm(config.n_embd)
self.mlp = nn.ModuleDict(dict(
c_fc = nn.Linear(config.n_embd, 4 * config.n_embd),
c_proj = nn.Linear(4 * config.n_embd, config.n_embd),
act = NewGELU(),
dropout = nn.Dropout(config.resid_pdrop),
))
m = self.mlp
self.mlpf = lambda x: m.dropout(m.c_proj(m.act(m.c_fc(x)))) # MLP forward
def forward(self, x):
x = x + self.attn(self.ln_1(x))
x = x + self.mlpf(self.ln_2(x))
return x
La clase GPT incluye la definición de varios tipos de modelos según los parámetros arquitectónicos¶
In [14]:
class GPT(nn.Module):
""" GPT Language Model """
@staticmethod
def get_default_config():
C = CN()
# either model_type or (n_layer, n_head, n_embd) must be given in the config
C.model_type = 'gpt'
C.n_layer = None
C.n_head = None
C.n_embd = None
# these options must be filled in externally
C.vocab_size = None
C.block_size = None
# dropout hyperparameters
C.embd_pdrop = 0.1
C.resid_pdrop = 0.1
C.attn_pdrop = 0.1
return C
def __init__(self, config):
super().__init__()
assert config.vocab_size is not None
assert config.block_size is not None
self.block_size = config.block_size
type_given = config.model_type is not None
params_given = all([config.n_layer is not None, config.n_head is not None, config.n_embd is not None])
assert type_given ^ params_given # exactly one of these (XOR)
if type_given:
# translate from model_type to detailed configuration
config.merge_from_dict({
# names follow the huggingface naming conventions
# GPT-1
'openai-gpt': dict(n_layer=12, n_head=12, n_embd=768), # 117M params
# GPT-2 configs
'gpt2': dict(n_layer=12, n_head=12, n_embd=768), # 124M params
'gpt2-medium': dict(n_layer=24, n_head=16, n_embd=1024), # 350M params
'gpt2-large': dict(n_layer=36, n_head=20, n_embd=1280), # 774M params
'gpt2-xl': dict(n_layer=48, n_head=25, n_embd=1600), # 1558M params
# Gophers
'gopher-44m': dict(n_layer=8, n_head=16, n_embd=512),
# (there are a number more...)
# I made these tiny models up
'gpt-mini': dict(n_layer=6, n_head=6, n_embd=192),
'gpt-micro': dict(n_layer=4, n_head=4, n_embd=128),
'gpt-nano': dict(n_layer=3, n_head=3, n_embd=48),
}[config.model_type])
self.transformer = nn.ModuleDict(dict(
wte = nn.Embedding(config.vocab_size, config.n_embd),
wpe = nn.Embedding(config.block_size, config.n_embd),
drop = nn.Dropout(config.embd_pdrop),
h = nn.ModuleList([Block(config) for _ in range(config.n_layer)]),
ln_f = nn.LayerNorm(config.n_embd),
))
self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
# init all weights, and apply a special scaled init to the residual projections, per GPT-2 paper
self.apply(self._init_weights)
for pn, p in self.named_parameters():
if pn.endswith('c_proj.weight'):
torch.nn.init.normal_(p, mean=0.0, std=0.02/math.sqrt(2 * config.n_layer))
# report number of parameters (note we don't count the decoder parameters in lm_head)
n_params = sum(p.numel() for p in self.transformer.parameters())
print("number of parameters: %.2fM" % (n_params/1e6,))
def _init_weights(self, module):
if isinstance(module, nn.Linear):
torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
if module.bias is not None:
torch.nn.init.zeros_(module.bias)
elif isinstance(module, nn.Embedding):
torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
elif isinstance(module, nn.LayerNorm):
torch.nn.init.zeros_(module.bias)
torch.nn.init.ones_(module.weight)
@classmethod
def from_pretrained(cls, model_type):
"""
Initialize a pretrained GPT model by copying over the weights
from a huggingface/transformers checkpoint.
"""
assert model_type in {'gpt2', 'gpt2-medium', 'gpt2-large', 'gpt2-xl'}
from transformers import GPT2LMHeadModel
# create a from-scratch initialized minGPT model
config = cls.get_default_config()
config.model_type = model_type
config.vocab_size = 50257 # openai's model vocabulary
config.block_size = 1024 # openai's model block_size
model = GPT(config)
sd = model.state_dict()
# init a huggingface/transformers model
model_hf = GPT2LMHeadModel.from_pretrained(model_type)
sd_hf = model_hf.state_dict()
# copy while ensuring all of the parameters are aligned and match in names and shapes
keys = [k for k in sd_hf if not k.endswith('attn.masked_bias')] # ignore these
transposed = ['attn.c_attn.weight', 'attn.c_proj.weight', 'mlp.c_fc.weight', 'mlp.c_proj.weight']
# basically the openai checkpoints use a "Conv1D" module, but we only want to use a vanilla nn.Linear.
# this means that we have to transpose these weights when we import them
assert len(keys) == len(sd)
for k in keys:
if any(k.endswith(w) for w in transposed):
# special treatment for the Conv1D weights we need to transpose
assert sd_hf[k].shape[::-1] == sd[k].shape
with torch.no_grad():
sd[k].copy_(sd_hf[k].t())
else:
# vanilla copy over the other parameters
assert sd_hf[k].shape == sd[k].shape
with torch.no_grad():
sd[k].copy_(sd_hf[k])
return model
def configure_optimizers(self, train_config):
"""
This long function is unfortunately doing something very simple and is being very defensive:
We are separating out all parameters of the model into two buckets: those that will experience
weight decay for regularization and those that won't (biases, and layernorm/embedding weights).
We are then returning the PyTorch optimizer object.
"""
# separate out all parameters to those that will and won't experience regularizing weight decay
decay = set()
no_decay = set()
whitelist_weight_modules = (torch.nn.Linear, )
blacklist_weight_modules = (torch.nn.LayerNorm, torch.nn.Embedding)
for mn, m in self.named_modules():
for pn, p in m.named_parameters():
fpn = '%s.%s' % (mn, pn) if mn else pn # full param name
# random note: because named_modules and named_parameters are recursive
# we will see the same tensors p many many times. but doing it this way
# allows us to know which parent module any tensor p belongs to...
if pn.endswith('bias'):
# all biases will not be decayed
no_decay.add(fpn)
elif pn.endswith('weight') and isinstance(m, whitelist_weight_modules):
# weights of whitelist modules will be weight decayed
decay.add(fpn)
elif pn.endswith('weight') and isinstance(m, blacklist_weight_modules):
# weights of blacklist modules will NOT be weight decayed
no_decay.add(fpn)
# validate that we considered every parameter
param_dict = {pn: p for pn, p in self.named_parameters()}
inter_params = decay & no_decay
union_params = decay | no_decay
assert len(inter_params) == 0, "parameters %s made it into both decay/no_decay sets!" % (str(inter_params), )
assert len(param_dict.keys() - union_params) == 0, "parameters %s were not separated into either decay/no_decay set!" \
% (str(param_dict.keys() - union_params), )
# create the pytorch optimizer object
optim_groups = [
{"params": [param_dict[pn] for pn in sorted(list(decay))], "weight_decay": train_config.weight_decay},
{"params": [param_dict[pn] for pn in sorted(list(no_decay))], "weight_decay": 0.0},
]
optimizer = torch.optim.AdamW(optim_groups, lr=train_config.learning_rate, betas=train_config.betas)
return optimizer
def forward(self, idx, targets=None):
device = idx.device
b, t = idx.size()
assert t <= self.block_size, f"Cannot forward sequence of length {t}, block size is only {self.block_size}"
pos = torch.arange(0, t, dtype=torch.long, device=device).unsqueeze(0) # shape (1, t)
# forward the GPT model itself
tok_emb = self.transformer.wte(idx) # token embeddings of shape (b, t, n_embd)
pos_emb = self.transformer.wpe(pos) # position embeddings of shape (1, t, n_embd)
x = self.transformer.drop(tok_emb + pos_emb)
for block in self.transformer.h:
x = block(x)
x = self.transformer.ln_f(x)
logits = self.lm_head(x)
# if we are given some desired targets also calculate the loss
loss = None
if targets is not None:
loss = F.cross_entropy(logits.view(-1, logits.size(-1)), targets.view(-1), ignore_index=-1)
return logits, loss
@torch.no_grad()
def generate(self, idx, max_new_tokens, temperature=1.0, do_sample=False, top_k=None):
"""
Take a conditioning sequence of indices idx (LongTensor of shape (b,t)) and complete
the sequence max_new_tokens times, feeding the predictions back into the model each time.
Most likely you'll want to make sure to be in model.eval() mode of operation for this.
"""
for _ in range(max_new_tokens):
# if the sequence context is growing too long we must crop it at block_size
idx_cond = idx if idx.size(1) <= self.block_size else idx[:, -self.block_size:]
# forward the model to get the logits for the index in the sequence
logits, _ = self(idx_cond)
# pluck the logits at the final step and scale by desired temperature
logits = logits[:, -1, :] / temperature
# optionally crop the logits to only the top k options
if top_k is not None:
v, _ = torch.topk(logits, top_k)
logits[logits < v[:, [-1]]] = -float('Inf')
# apply softmax to convert logits to (normalized) probabilities
probs = F.softmax(logits, dim=-1)
# either sample from the distribution or take the most likely element
if do_sample:
idx_next = torch.multinomial(probs, num_samples=1)
else:
_, idx_next = torch.topk(probs, k=1, dim=-1)
# append sampled index to the running sequence and continue
idx = torch.cat((idx, idx_next), dim=1)
return idx
El trainer implementa el forward sobre los datos de train.¶
In [6]:
import time
from collections import defaultdict
from torch.utils.data.dataloader import DataLoader
class Trainer:
@staticmethod
def get_default_config():
C = CN()
# device to train on
#C.device = 'auto'
C.device = 'cpu'
# dataloder parameters
C.num_workers = 4
# optimizer parameters
C.max_iters = None
C.batch_size = 64
C.learning_rate = 3e-4
C.betas = (0.9, 0.95)
C.weight_decay = 0.1
C.grad_norm_clip = 1.0
return C
def __init__(self, config, model, train_dataset):
self.config = config
self.model = model
self.optimizer = None
self.train_dataset = train_dataset
self.callbacks = defaultdict(list)
# determine the device we'll train on
if config.device == 'auto':
self.device = 'cuda' if torch.cuda.is_available() else 'cpu'
else:
self.device = config.device
self.model = self.model.to(self.device)
print("running on device", self.device)
# variables that will be assigned to trainer class later for logging and etc
self.iter_num = 0
self.iter_time = 0.0
self.iter_dt = 0.0
def add_callback(self, onevent: str, callback):
self.callbacks[onevent].append(callback)
def set_callback(self, onevent: str, callback):
self.callbacks[onevent] = [callback]
def trigger_callbacks(self, onevent: str):
for callback in self.callbacks.get(onevent, []):
callback(self)
def run(self):
model, config = self.model, self.config
# setup the optimizer
self.optimizer = model.configure_optimizers(config)
# setup the dataloader
train_loader = DataLoader(
self.train_dataset,
sampler=torch.utils.data.RandomSampler(self.train_dataset, replacement=True, num_samples=int(1e10)),
shuffle=False,
pin_memory=True,
batch_size=config.batch_size,
num_workers=config.num_workers,
)
model.train()
self.iter_num = 0
self.iter_time = time.time()
data_iter = iter(train_loader)
while True:
# fetch the next batch (x, y) and re-init iterator if needed
try:
batch = next(data_iter)
except StopIteration:
data_iter = iter(train_loader)
batch = next(data_iter)
batch = [t.to(self.device) for t in batch]
x, y = batch
# forward the model
logits, self.loss = model(x, y)
# backprop and update the parameters
model.zero_grad(set_to_none=True)
self.loss.backward()
torch.nn.utils.clip_grad_norm_(model.parameters(), config.grad_norm_clip)
self.optimizer.step()
self.trigger_callbacks('on_batch_end')
self.iter_num += 1
tnow = time.time()
self.iter_dt = tnow - self.iter_time
self.iter_time = tnow
# termination conditions
if config.max_iters is not None and self.iter_num >= config.max_iters:
break
En este ejemplo, la tarea consiste en ordenar una secuencia de enteros. Esto tiene particularidades ya que usaremos una máscara 'I' para indicar al transformer que estamos ingresando la entrada, estrategia denominada prefix tuning.¶
In [7]:
from torch.utils.data import Dataset
from torch.utils.data.dataloader import DataLoader
import pickle
class SortDataset(Dataset):
"""
Dataset for the Sort problem. E.g. for problem length 6:
Input: 0 0 2 1 0 1 -> Output: 0 0 0 1 1 2
Which will feed into the transformer concatenated as:
input: 0 0 2 1 0 1 0 0 0 1 1
output: I I I I I 0 0 0 1 1 2
where I is "ignore", as the transformer is reading the input sequence
"""
def __init__(self, split, length=6, num_digits=3):
assert split in {'train', 'test'}
self.split = split
self.length = length
self.num_digits = num_digits
def __len__(self):
return 10000 # ...
def get_vocab_size(self):
return self.num_digits
def get_block_size(self):
# the length of the sequence that will feed into transformer,
# containing concatenated input and the output, but -1 because
# the transformer starts making predictions at the last input element
return self.length * 2 - 1
def __getitem__(self, idx):
# use rejection sampling to generate an input example from the desired split
while True:
# generate some random integers
inp = torch.randint(self.num_digits, size=(self.length,), dtype=torch.long)
# half of the time let's try to boost the number of examples that
# have a large number of repeats, as this is what the model seems to struggle
# with later in training, and they are kind of rate
if torch.rand(1).item() < 0.5:
if inp.unique().nelement() > self.length // 2:
# too many unqiue digits, re-sample
continue
# figure out if this generated example is train or test based on its hash
h = hash(pickle.dumps(inp.tolist()))
inp_split = 'test' if h % 4 == 0 else 'train' # designate 25% of examples as test
if inp_split == self.split:
break # ok
# solve the task: i.e. sort
sol = torch.sort(inp)[0]
# concatenate the problem specification and the solution
cat = torch.cat((inp, sol), dim=0)
# the inputs to the transformer will be the offset sequence
x = cat[:-1].clone()
y = cat[1:].clone()
# we only want to predict at output locations, mask out the loss at the input locations
y[:self.length-1] = -1
return x, y
In [8]:
train_dataset = SortDataset('train')
test_dataset = SortDataset('test')
x, y = train_dataset[0]
for a, b in zip(x,y):
print(int(a),int(b))
2 -1 1 -1 0 -1 0 -1 2 -1 2 0 0 0 0 1 1 2 2 2 2 2
Este es un modelo pequeño, ya que lo vamos a entrenar en cpu para simplificar el tema de acceso a gpus¶
In [9]:
model_config = GPT.get_default_config()
model_config.model_type = 'gpt-nano'
model_config.vocab_size = train_dataset.get_vocab_size()
model_config.block_size = train_dataset.get_block_size()
model = GPT(model_config)
number of parameters: 0.09M
In [10]:
CUDA_LAUNCH_BLOCKING=1
train_config = Trainer.get_default_config()
train_config.learning_rate = 5e-4
train_config.max_iters = 10000
train_config.num_workers = 0
trainer = Trainer(train_config, model, train_dataset)
running on device cpu
In [11]:
def batch_end_callback(trainer):
if trainer.iter_num % 100 == 0:
print(f"iter_dt {trainer.iter_dt * 1000:.2f}ms; iter {trainer.iter_num}: train loss {trainer.loss.item():.5f}")
trainer.set_callback('on_batch_end', batch_end_callback)
trainer.run()
iter_dt 0.00ms; iter 0: train loss 1.14617 iter_dt 16.91ms; iter 100: train loss 0.16010 iter_dt 22.04ms; iter 200: train loss 0.06826 iter_dt 15.27ms; iter 300: train loss 0.06172 iter_dt 22.94ms; iter 400: train loss 0.04327 iter_dt 17.05ms; iter 500: train loss 0.02389 iter_dt 15.96ms; iter 600: train loss 0.02782 iter_dt 17.86ms; iter 700: train loss 0.00776 iter_dt 17.04ms; iter 800: train loss 0.01967 iter_dt 26.84ms; iter 900: train loss 0.01876 iter_dt 16.68ms; iter 1000: train loss 0.02067 iter_dt 15.81ms; iter 1100: train loss 0.01211 iter_dt 16.99ms; iter 1200: train loss 0.01564 iter_dt 16.60ms; iter 1300: train loss 0.00063 iter_dt 16.71ms; iter 1400: train loss 0.00075 iter_dt 16.56ms; iter 1500: train loss 0.00899 iter_dt 17.18ms; iter 1600: train loss 0.01194 iter_dt 17.26ms; iter 1700: train loss 0.01115 iter_dt 16.26ms; iter 1800: train loss 0.00106 iter_dt 15.51ms; iter 1900: train loss 0.01863 iter_dt 18.32ms; iter 2000: train loss 0.00370 iter_dt 16.28ms; iter 2100: train loss 0.00263 iter_dt 16.87ms; iter 2200: train loss 0.00784 iter_dt 17.85ms; iter 2300: train loss 0.00167 iter_dt 16.53ms; iter 2400: train loss 0.00545 iter_dt 24.46ms; iter 2500: train loss 0.00599 iter_dt 17.15ms; iter 2600: train loss 0.00060 iter_dt 15.42ms; iter 2700: train loss 0.00168 iter_dt 16.85ms; iter 2800: train loss 0.00045 iter_dt 26.57ms; iter 2900: train loss 0.02349 iter_dt 16.13ms; iter 3000: train loss 0.00041 iter_dt 16.44ms; iter 3100: train loss 0.00121 iter_dt 16.10ms; iter 3200: train loss 0.02541 iter_dt 21.17ms; iter 3300: train loss 0.02491 iter_dt 16.09ms; iter 3400: train loss 0.00166 iter_dt 23.06ms; iter 3500: train loss 0.01533 iter_dt 23.91ms; iter 3600: train loss 0.00018 iter_dt 25.28ms; iter 3700: train loss 0.00279 iter_dt 17.21ms; iter 3800: train loss 0.00076 iter_dt 15.68ms; iter 3900: train loss 0.00276 iter_dt 25.46ms; iter 4000: train loss 0.00343 iter_dt 14.45ms; iter 4100: train loss 0.00077 iter_dt 17.21ms; iter 4200: train loss 0.00149 iter_dt 16.79ms; iter 4300: train loss 0.00072 iter_dt 15.74ms; iter 4400: train loss 0.01602 iter_dt 17.56ms; iter 4500: train loss 0.00027 iter_dt 16.34ms; iter 4600: train loss 0.00031 iter_dt 16.31ms; iter 4700: train loss 0.00040 iter_dt 19.95ms; iter 4800: train loss 0.00219 iter_dt 19.18ms; iter 4900: train loss 0.04403 iter_dt 28.14ms; iter 5000: train loss 0.00515 iter_dt 30.79ms; iter 5100: train loss 0.00121 iter_dt 20.64ms; iter 5200: train loss 0.00155 iter_dt 24.16ms; iter 5300: train loss 0.00013 iter_dt 16.49ms; iter 5400: train loss 0.00057 iter_dt 17.62ms; iter 5500: train loss 0.00095 iter_dt 16.52ms; iter 5600: train loss 0.00009 iter_dt 21.65ms; iter 5700: train loss 0.01718 iter_dt 19.77ms; iter 5800: train loss 0.00049 iter_dt 23.02ms; iter 5900: train loss 0.00028 iter_dt 16.60ms; iter 6000: train loss 0.00568 iter_dt 16.45ms; iter 6100: train loss 0.00390 iter_dt 29.99ms; iter 6200: train loss 0.02280 iter_dt 16.08ms; iter 6300: train loss 0.00080 iter_dt 16.39ms; iter 6400: train loss 0.00013 iter_dt 17.62ms; iter 6500: train loss 0.01391 iter_dt 20.30ms; iter 6600: train loss 0.00114 iter_dt 21.15ms; iter 6700: train loss 0.00005 iter_dt 20.87ms; iter 6800: train loss 0.00590 iter_dt 17.20ms; iter 6900: train loss 0.00297 iter_dt 30.51ms; iter 7000: train loss 0.00230 iter_dt 16.80ms; iter 7100: train loss 0.00382 iter_dt 16.53ms; iter 7200: train loss 0.00147 iter_dt 27.03ms; iter 7300: train loss 0.00871 iter_dt 20.12ms; iter 7400: train loss 0.01805 iter_dt 16.47ms; iter 7500: train loss 0.01507 iter_dt 15.83ms; iter 7600: train loss 0.00013 iter_dt 15.69ms; iter 7700: train loss 0.00010 iter_dt 15.69ms; iter 7800: train loss 0.00023 iter_dt 15.49ms; iter 7900: train loss 0.00037 iter_dt 27.93ms; iter 8000: train loss 0.00015 iter_dt 15.68ms; iter 8100: train loss 0.00082 iter_dt 16.84ms; iter 8200: train loss 0.00012 iter_dt 17.36ms; iter 8300: train loss 0.00007 iter_dt 16.18ms; iter 8400: train loss 0.00006 iter_dt 16.50ms; iter 8500: train loss 0.00010 iter_dt 17.78ms; iter 8600: train loss 0.00021 iter_dt 17.23ms; iter 8700: train loss 0.00011 iter_dt 17.15ms; iter 8800: train loss 0.00048 iter_dt 16.67ms; iter 8900: train loss 0.00545 iter_dt 16.85ms; iter 9000: train loss 0.00007 iter_dt 16.48ms; iter 9100: train loss 0.00003 iter_dt 16.49ms; iter 9200: train loss 0.00015 iter_dt 15.75ms; iter 9300: train loss 0.00626 iter_dt 20.84ms; iter 9400: train loss 0.00088 iter_dt 17.90ms; iter 9500: train loss 0.00047 iter_dt 16.40ms; iter 9600: train loss 0.00042 iter_dt 31.59ms; iter 9700: train loss 0.00012 iter_dt 18.27ms; iter 9800: train loss 0.00790 iter_dt 26.44ms; iter 9900: train loss 0.00005
En eval split comparamos la secuencia entregada por gpt con la secuencia ordenada y medimos los errores.¶
In [13]:
def eval_split(trainer, split, max_batches):
dataset = {'train':train_dataset, 'test':test_dataset}[split]
n = train_dataset.length
print(n)
results = []
mistakes_printed_already = 0
loader = DataLoader(dataset, batch_size=100, num_workers=0, drop_last=False)
for b, (x, y) in enumerate(loader):
x = x.to(trainer.device)
y = y.to(trainer.device)
# isolate the input pattern alone
inp = x[:, :n]
sol = y[:, -n:]
# let the model sample the rest of the sequence
cat = model.generate(inp, n, do_sample=False) # using greedy argmax, not sampling
sol_candidate = cat[:, n:] # isolate the filled in sequence
# compare the predicted sequence to the true sequence
correct = (sol == sol_candidate).all(1).cpu()
for i in range(x.size(0)):
results.append(int(correct[i]))
if not correct[i] and mistakes_printed_already < 3: # only print up to 5 mistakes to get a sense
mistakes_printed_already += 1
print("GPT claims that %s sorted is %s but gt is %s" % (inp[i].tolist(), sol_candidate[i].tolist(), sol[i].tolist()))
if max_batches is not None and b+1 >= max_batches:
break
rt = torch.tensor(results, dtype=torch.float)
print("%s final score: %d/%d = %.2f%% correct" % (split, rt.sum(), len(results), 100*rt.mean()))
return rt.sum()
# run a lot of examples from both train and test through the model and verify the output correctness
with torch.no_grad():
train_score = eval_split(trainer, 'train', max_batches=50)
test_score = eval_split(trainer, 'test', max_batches=50)
6 GPT claims that [0, 1, 0, 1, 2, 1] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2] GPT claims that [0, 2, 1, 1, 0, 1] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2] GPT claims that [0, 1, 1, 1, 0, 2] sorted is [0, 1, 1, 1, 1, 2] but gt is [0, 0, 1, 1, 1, 2] train final score: 4977/5000 = 99.54% correct 6 GPT claims that [0, 1, 1, 2, 0, 0] sorted is [0, 0, 1, 1, 1, 2] but gt is [0, 0, 0, 1, 1, 2] GPT claims that [1, 2, 2, 1, 2, 2] sorted is [1, 2, 2, 2, 2, 2] but gt is [1, 1, 2, 2, 2, 2] GPT claims that [2, 2, 2, 1, 2, 2] sorted is [2, 2, 2, 2, 2, 2] but gt is [1, 2, 2, 2, 2, 2] test final score: 4987/5000 = 99.74% correct
In [13]:
n = train_dataset.length
inp = torch.tensor([[0, 0, 2, 1, 0, 1]], dtype=torch.long).to(trainer.device)
assert inp[0].nelement() == n
with torch.no_grad():
cat = model.generate(inp, n, do_sample=False)
sol = torch.sort(inp[0])[0]
sol_candidate = cat[:, n:]
print('input sequence :', inp.tolist())
print('predicted sorted:', sol_candidate.tolist())
print('gt sort :', sol.tolist())
print('matches :', bool((sol == sol_candidate).all()))
input sequence : [[0, 0, 2, 1, 0, 1]] predicted sorted: [[0, 0, 0, 1, 1, 2]] gt sort : [0, 0, 0, 1, 1, 2] matches : True
In [ ]: