# Supervised Fine-Tuning

Fine-tune GPT2 for Question Answering Task

see: https://www.kaggle.com/code/dsmeena/pytorch-fine-tuning-gpt2-for-questionanswering


In [4]:
# !pip install datasets, torch, transformers

In [1]:
import torch

## Load the Quetsion & Answer squad Dataset from Pytorch

https://pytorch.org/text/stable/datasets.html#squad-1-0

In [23]:
from torch.utils.data import Dataset
from datasets import load_dataset

train_dataset = load_dataset("squad")["train"]
print(len(train_dataset))
test_dataset = load_dataset("squad")["validation"]
print(len(test_dataset))

87599
10570


In [42]:
# get some elements of the dataset
generator = train_dataset.iter(batch_size=5)
y = 0
for x in generator:
    print(x.keys())
    questions = x['question']
    answers = x['answers']
    for q, a in zip(questions, answers):
        print("Q", q)
        print("A", a)
    break

dict_keys(['id', 'title', 'context', 'question', 'answers'])
Q To whom did the Virgin Mary allegedly appear in 1858 in Lourdes France?
A {'text': ['Saint Bernadette Soubirous'], 'answer_start': [515]}
Q What is in front of the Notre Dame Main Building?
A {'text': ['a copper statue of Christ'], 'answer_start': [188]}
Q The Basilica of the Sacred heart at Notre Dame is beside to which structure?
A {'text': ['the Main Building'], 'answer_start': [279]}
Q What is the Grotto at Notre Dame?
A {'text': ['a Marian place of prayer and reflection'], 'answer_start': [381]}
Q What sits on top of the Main Building at Notre Dame?
A {'text': ['a golden statue of the Virgin Mary'], 'answer_start': [92]}


In [43]:
test_dataset[0]

{'id': '56be4db0acb8001400a502ec',
 'title': 'Super_Bowl_50',
 'context': 'Super Bowl 50 was an American football game to determine the champion of the National Football League (NFL) for the 2015 season. The American Football Conference (AFC) champion Denver Broncos defeated the National Football Conference (NFC) champion Carolina Panthers 24–10 to earn their third Super Bowl title. The game was played on February 7, 2016, at Levi\'s Stadium in the San Francisco Bay Area at Santa Clara, California. As this was the 50th Super Bowl, the league emphasized the "golden anniversary" with various gold-themed initiatives, as well as temporarily suspending the tradition of naming each Super Bowl game with Roman numerals (under which the game would have been known as "Super Bowl L"), so that the logo could prominently feature the Arabic numerals 50.',
 'question': 'Which NFL team represented the AFC at Super Bowl 50?',
 'answers': {'text': ['Denver Broncos', 'Denver Broncos', 'Denver Broncos'],


## Build a custom dataset out of squad

In [44]:
class SquadDataset(Dataset):
    def __init__(self, orig_dataset, tokenizer):
        self.tokenizer = tokenizer
        self.dataset = orig_dataset
        
    def __len__(self):
        return len(self.dataset)
    
    def __getitem__(self, idx):
        item = self.dataset[idx]
        context = item['context']
        question = item['question']
        answer = item['answers']['text'][0] # get the first answer from answers
        
        # do encoding of the context and question 
        encoding = self.tokenizer.encode_plus(
            question,
            context,
            add_special_tokens=True,
            return_token_type_ids=True,
            return_attention_mask=True,
            padding='max_length',   
            max_length=384,     #max prompt size of GPT2
            truncation=True
        )
        
        # get start and end positions of answer in input_ids
        input_ids = encoding['input_ids']
        answer_start = item['answers']['answer_start'][0]
        answer_end = answer_start + len(answer)
        
        start_positions = []
        end_positions = []
        for i, token_id in enumerate(input_ids):
            if i == answer_start:
                start_positions.append(i)
            else:
                start_positions.append(-100)
            
            if i == answer_end:
                end_positions.append(i)
            else:
                end_positions.append(-100)
        
        # Create input tensors
        inputs = {
            'input_ids': torch.tensor(encoding['input_ids'], dtype=torch.long),
            'attention_mask': torch.tensor(encoding['attention_mask'], dtype=torch.long),
            'token_type_ids': torch.tensor(encoding['token_type_ids'], dtype=torch.long),
            'start_positions': torch.tensor(start_positions, dtype=torch.float),  # start and end positions should be float
            'end_positions': torch.tensor(end_positions, dtype=torch.float)
        }
        
        return inputs, answer

In [45]:
from transformers import GPT2Tokenizer
tokenizer = GPT2Tokenizer.from_pretrained('gpt2')
tokenizer.pad_token = tokenizer.eos_token



In [46]:
train = SquadDataset(train_dataset, tokenizer)

In [57]:
len(train)
inputs , answer = train[0]
input_ids = inputs['input_ids']
print("Question & Context:")
print(tokenizer.decode(input_ids))
print("Answer:")
print(answer)

Question & Context:
To whom did the Virgin Mary allegedly appear in 1858 in Lourdes France?Architecturally, the school has a Catholic character. Atop the Main Building's gold dome is a golden statue of the Virgin Mary. Immediately in front of the Main Building and facing it, is a copper statue of Christ with arms upraised with the legend "Venite Ad Me Omnes". Next to the Main Building is the Basilica of the Sacred Heart. Immediately behind the basilica is the Grotto, a Marian place of prayer and reflection. It is a replica of the grotto at Lourdes, France where the Virgin Mary reputedly appeared to Saint Bernadette Soubirous in 1858. At the end of the main drive (and in a direct line that connects through 3 statues and the Gold Dome), is a simple, modern stone statue of Mary.<|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endoftext|><|endo

In [58]:
# Use torch DataLoader
from torch.utils.data import DataLoader

# Create the Dataloader
train_dataloader = DataLoader(
    SquadDataset(train_dataset, tokenizer),
    batch_size=16,
    shuffle=True
)
test_dataloader = DataLoader(
    SquadDataset(test_dataset, tokenizer),
    batch_size=16,
    shuffle=True
)

In [62]:
# Iterate through DataLoader

for batch in train_dataloader:  
    print(len(batch[0]['input_ids'])) #each batch contains 16 input_ids
    print(f"Our context is:\n {batch[0]['input_ids']}")
    break

16
Our context is:
 tensor([[ 2061,   614,   373,  ..., 50256, 50256, 50256],
        [ 2061, 22987,  6903,  ..., 50256, 50256, 50256],
        [ 2061,  3858,   286,  ..., 50256, 50256, 50256],
        ...,
        [ 2061,   466, 31877,  ..., 50256, 50256, 50256],
        [ 2061,   547,   973,  ..., 50256, 50256, 50256],
        [ 8496,  2073,  2098,  ..., 50256, 50256, 50256]])


In [63]:
device = (
    "cuda"
    if torch.cuda.is_available()
    else "mps"
    if torch.backends.mps.is_available()
    else "cpu"
)
print(f"Using {device} device")

Using mps device


In [91]:
######## Load GPT2
#from transformers import AutoModel
#model = AutoModel.from_pretrained("gpt2").to(device)
#print(model)
#from transformers import AutoModelWithLMHead
#model = AutoModelWithLMHead.from_pretrained("gpt2").to(device)
#print(model)

In [92]:
from transformers import AutoModelForQuestionAnswering
model = AutoModelForQuestionAnswering.from_pretrained("gpt2").to(device)
print(model)

Some weights of GPT2ForQuestionAnswering were not initialized from the model checkpoint at gpt2 and are newly initialized: ['qa_outputs.bias', 'qa_outputs.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


GPT2ForQuestionAnswering(
  (transformer): GPT2Model(
    (wte): Embedding(50257, 768)
    (wpe): Embedding(1024, 768)
    (drop): Dropout(p=0.1, inplace=False)
    (h): ModuleList(
      (0-11): 12 x GPT2Block(
        (ln_1): LayerNorm((768,), eps=1e-05, elementwise_affine=True)
        (attn): GPT2SdpaAttention(
          (c_attn): Conv1D(nf=2304, nx=768)
          (c_proj): Conv1D(nf=768, nx=768)
          (attn_dropout): Dropout(p=0.1, inplace=False)
          (resid_dropout): Dropout(p=0.1, inplace=False)
        )
        (ln_2): LayerNorm((768,), eps=1e-05, elementwise_affine=True)
        (mlp): GPT2MLP(
          (c_fc): Conv1D(nf=3072, nx=768)
          (c_proj): Conv1D(nf=768, nx=3072)
          (act): NewGELUActivation()
          (dropout): Dropout(p=0.1, inplace=False)
        )
      )
    )
    (ln_f): LayerNorm((768,), eps=1e-05, elementwise_affine=True)
  )
  (qa_outputs): Linear(in_features=768, out_features=2, bias=True)
)


In [95]:
# Hyperparameters
learning_rate = 5e-5
epochs = 5

In [96]:
from transformers import AdamW

optimizer = AdamW(model.parameters(), lr=learning_rate)
# loss_fn = model.get_loss()

In [None]:
def train_loop(dataloader, model, optimizer):
    
    # set the model to training model
    model.train()
    
    for batch in dataloader:
        optimizer.zero_grad()
        
        # previous tokens
        input_ids = batch[0]['input_ids'].to(device)
        attention_mask = batch[0]['attention_mask'].to(device)
        token_type_ids = batch[0]['token_type_ids'].to(device)
        start_positions = batch[0]['start_positions'].to(device)
        end_positions = batch[0]['end_positions'].to(device)
        
        labels = {
            'start_positions': start_positions,
            'end_positions': end_positions
        }
        
       # get outputs from model
        outputs = model(input_ids, attention_mask=attention_mask, token_type_ids=token_type_ids)

        # calculate loss
        loss_start = nn.CrossEntropyLoss()(outputs.start_logits, start_positions)
        loss_end = nn.CrossEntropyLoss()(outputs.end_logits, end_positions)
        loss = (loss_start + loss_end) / 2  # average loss for start and end positions
        
        # backpropagation
        loss.backward()
        optimizer.step()
        

def test_loop(dataloader, model):
    # set the model of evaluation
    model.eval()
    val_loss = 0
    
    # Evaluating the model with torch.no_grad() ensures that no gradients are computed during test mode
    with torch.no_grad():
        for batch in dataloader:
            # previous tokens
            input_ids = batch[0]['input_ids'].to(device)
            attention_mask = batch[0]['attention_mask'].to(device)
            token_type_ids = batch[0]['token_type_ids'].to(device)
            start_positions = batch[0]['start_positions'].to(device)
            end_positions = batch[0]['end_positions'].to(device)

            labels = {
                'start_positions': start_positions,
                'end_positions': end_positions
            }

           # get outputs from model
            outputs = model(input_ids, attention_mask=attention_mask, token_type_ids=token_type_ids)

            # calculate loss
            loss_start = nn.CrossEntropyLoss()(outputs.start_logits, start_positions)
            loss_end = nn.CrossEntropyLoss()(outputs.end_logits, end_positions)
            loss = (loss_start + loss_end) / 2  # average loss for start and end positions
            
            val_loss += loss.item()
    
    # Print the validation loss for this epoch
    print(f"Validation Loss: {val_loss/len(dataloader)}")


In [None]:
import transformers
import torch.nn as nn
transformers.logging.set_verbosity_error()

for t in range(epochs):
    print(f"Epoch {t+1}\n ---------------------------")
    train_loop(train_dataloader, model, optimizer)
    test_loop(test_dataloader, model)

print("Done!")


# model returns a CausalLMOutputWithCrossAttentions object, not just a loss. We can get the loss using loss attribute on it.
# This is the recommeded way of obtaining the loss value when using the transformers library.

In [None]:
# Save your fine-tuned model
model.save_pretrained("fine_tuned_QA")