In [8]:
import math
import re
from random import *
import numpy as np
import torch
import torch.nn as nn
import torch.optim as optim
import pandas as pd
from sklearn.model_selection import train_test_split

device = torch.device("cuda")

In [9]:
df = pd.read_csv("./out/labelled_dataset.csv")

df

Unnamed: 0,a[0],a[1],a[2],a[3],eigen_values,results,numeric_output
0,9,-2,2,-3,"['8.65685424949238', '-2.65685424949238']",1,"'[0],[0]'"
1,-9,-3,2,-4,"['-7.0', '-6.0']",0,"'[0],[0]'"
2,-5,-9,2,1,"['(-1.9999999999999998+3.000000000000001j)', '...",0,"'[0],[0]'"
3,-7,-7,-1,-5,"['-8.82842712474619', '-3.17157287525381']",0,"'[0],[0]'"
4,7,-7,-9,4,"['13.577747210701755', '-2.577747210701756']",1,"'[0],[0]'"
...,...,...,...,...,...,...,...
996,2,-6,8,-9,"['(-3.500000000000001+4.21307488658818j)', '(-...",0,"'[0],[0]'"
997,8,-9,4,4,"['(6+5.65685424949238j)', '(6-5.65685424949238...",1,"'[0],[0]'"
998,-3,-9,8,-9,"['(-6+7.937253933193773j)', '(-6-7.93725393319...",0,"'[0],[0]'"
999,-6,-1,4,-7,"['(-6.5+1.9364916731037083j)', '(-6.5-1.936491...",0,"'[0],[0]'"


In [18]:
features = ["numeric_output"]

X = df[features]
y = df.results

X, y

(     numeric_output
 0         '[0],[0]'
 1         '[0],[0]'
 2         '[0],[0]'
 3         '[0],[0]'
 4         '[0],[0]'
 ...             ...
 996       '[0],[0]'
 997       '[0],[0]'
 998       '[0],[0]'
 999       '[0],[0]'
 1000            NaN
 
 [1001 rows x 1 columns],
 0       1
 1       0
 2       0
 3       0
 4       1
        ..
 996     0
 997     1
 998     0
 999     0
 1000    1
 Name: results, Length: 1001, dtype: int64)

In [19]:
train_X, val_X, train_y, val_y = train_test_split(X, y)

In [2]:
text = (
        'Hello, how are you? I am Romeo.\n'
        'Hello, Romeo My name is Juliet. Nice to meet you.\n'
        'Nice meet you too. How are you today?\n'
        'Great. My baseball team won the competition.\n'
        'Oh Congratulations, Juliet\n'
        'Thank you Romeo'
    )

In [27]:
sentences = re.sub("[.,!?\\-]", '', text.lower()).split('\n')  
word_list = list(set(" ".join(sentences).split()))
word_dict = {'[PAD]': 0, '[CLS]': 1, '[SEP]': 2, '[MASK]': 3}


for i, w in enumerate(word_list):
    word_dict[w] = i + 4
number_dict = {i: w for i, w in enumerate(word_dict)}
vocab_size = len(word_dict)

token_list = list()
for sentence in sentences:
    arr = [word_dict[s] for s in sentence.split()]
    token_list.append(arr)

In [4]:
token_list

[[8, 4, 11, 9, 16, 27, 5],
 [8, 5, 6, 14, 17, 15, 20, 23, 12, 9],
 [20, 12, 9, 24, 4, 11, 9, 28],
 [21, 6, 19, 25, 18, 7, 13],
 [10, 26, 15],
 [22, 9, 5]]

In [5]:
maxlen = 30 
batch_size = 6
max_pred = 5  
n_layers = 6 
n_heads = 12 
d_model = 768 
d_ff = 768 * 4  
d_k = d_v = 64  
n_segments = 2

In [6]:

def make_batch():
    batch = []
    positive = negative = 0
    while positive != batch_size/2 or negative != batch_size/2:
        tokens_a_index, tokens_b_index= randrange(len(sentences)), randrange(len(sentences))
        tokens_a, tokens_b= token_list[tokens_a_index], token_list[tokens_b_index]

        input_ids = [word_dict['[CLS]']] + tokens_a + [word_dict['[SEP]']] + tokens_b + [word_dict['[SEP]']]

        segment_ids = [0] * (1 + len(tokens_a) + 1) + [1] * (len(tokens_b) + 1)

        n_pred =  min(max_pred, max(1, int(round(len(input_ids) * 0.15)))) 

        cand_maked_pos = [i for i, token in enumerate(input_ids)
                          if token != word_dict['[CLS]'] and token != word_dict['[SEP]']]
        shuffle(cand_maked_pos)
        masked_tokens, masked_pos = [], []
        for pos in cand_maked_pos[:n_pred]:
            masked_pos.append(pos)
            masked_tokens.append(input_ids[pos])
            if random() < 0.8:  
                input_ids[pos] = word_dict['[MASK]'] 
            elif random() < 0.5:  
                index = randint(0, vocab_size - 1) 
                input_ids[pos] = word_dict[number_dict[index]] 
        
        n_pad = maxlen - len(input_ids)
        input_ids.extend([0] * n_pad)
        segment_ids.extend([0] * n_pad)
    
        if max_pred > n_pred:
            n_pad = max_pred - n_pred
            masked_tokens.extend([0] * n_pad)
            masked_pos.extend([0] * n_pad)

        if tokens_a_index + 1 == tokens_b_index and positive < batch_size/2:
            batch.append([input_ids, segment_ids, masked_tokens, masked_pos, True]) 
            positive += 1
        elif tokens_a_index + 1 != tokens_b_index and negative < batch_size/2:
            batch.append([input_ids, segment_ids, masked_tokens, masked_pos, False]) 
            negative += 1
    return batch
        

In [7]:
def get_attn_pad_mask(seq_q, seq_k):
    batch_size, len_q = seq_q.size()
    batch_size, len_k = seq_k.size()
    pad_attn_mask = seq_k.data.eq(0).unsqueeze(1)
    pad_attn_mask = pad_attn_mask.to(device)
    return pad_attn_mask.expand(batch_size, len_q, len_k)  

In [8]:
def gelu(x):
    return x * 0.5 * (1.0 + torch.erf(x / math.sqrt(2.0)))

In [9]:
batch = make_batch()

In [10]:
input_ids, segment_ids, masked_tokens, masked_pos, isNext = map(torch.LongTensor, zip(*batch))

#########################################
input_ids = input_ids.to(device)
segment_ids = segment_ids.to(device)
masked_tokens = masked_tokens.to(device)
masked_pos = masked_pos.to(device)
isNext = isNext.to(device)

In [11]:
get_attn_pad_mask(input_ids, input_ids)[0][0], input_ids[0], segment_ids[0], masked_tokens[0], masked_pos[0], isNext[0]

(tensor([False, False, False, False, False, False, False, False, False, False,
         False, False, False, False, False, False,  True,  True,  True,  True,
          True,  True,  True,  True,  True,  True,  True,  True,  True,  True],
        device='cuda:0'),
 tensor([ 1, 22,  9,  5,  2,  8,  3,  6, 14, 17, 15, 20, 23,  3,  9,  2,  0,  0,
          0,  0,  0,  0,  0,  0,  0,  0,  0,  0,  0,  0], device='cuda:0'),
 tensor([0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0,
         0, 0, 0, 0, 0, 0], device='cuda:0'),
 tensor([12,  5,  0,  0,  0], device='cuda:0'),
 tensor([13,  6,  0,  0,  0], device='cuda:0'),
 tensor(0, device='cuda:0'))

In [12]:
class Embedding(nn.Module):
    def __init__(self):
        super(Embedding, self).__init__()
        self.tok_embed = nn.Embedding(vocab_size, d_model)
        self.pos_embed = nn.Embedding(maxlen, d_model)
        self.seg_embed = nn.Embedding(n_segments, d_model)
        self.norm = nn.LayerNorm(d_model)

    def forward(self, x, seg):
        seq_len = x.size(1)
        pos = torch.arange(seq_len, dtype=torch.long)
        pos = pos.unsqueeze(0).expand_as(x)
        x = x.to(device)
        pos = pos.to(device)
        seg = seg.to(device)
        embedding = self.tok_embed(x) + self.pos_embed(pos) + self.seg_embed(seg)
        embedding = embedding.to(device)
        return self.norm(embedding)

In [13]:
class ScaledDotProductAttention(nn.Module):
    def __init__(self):
        super(ScaledDotProductAttention, self).__init__()

    def forward(self, Q, K, V, attn_mask):
        scores = torch.matmul(Q, K.transpose(-1, -2)) / np.sqrt(d_k)
        scores.masked_fill_(attn_mask, -1e9)
        scores = scores.to(device)
        attn = nn.Softmax(dim=-1)(scores)
        attn = attn.to(device)
        context = torch.matmul(attn, V)
        context = context.to(device)
        return scores, context, attn 

In [14]:
emb = Embedding()
emb = emb.to(device)
input_ids = input_ids.to(device)
segment_ids = segment_ids.to(device)
embeds = emb(input_ids, segment_ids)

attenM = get_attn_pad_mask(input_ids, input_ids)

attenM = attenM.to(device)

SDPA= ScaledDotProductAttention()(embeds, embeds, embeds, attenM)

S, C, A = SDPA

S = S.to(device)
C = C.to(device)
A = A.to(device)


print('Masks',attenM[0][0])
print()
print('Scores: ', S[0][0],'\n\nAttention M: ', A[0][0])

Masks tensor([False, False, False, False, False, False, False, False, False, False,
        False, False, False, False, False, False,  True,  True,  True,  True,
         True,  True,  True,  True,  True,  True,  True,  True,  True,  True],
       device='cuda:0')

Scores:  tensor([ 9.6000e+01,  3.4575e+01,  3.4091e+01,  3.3278e+01,  2.8368e+01,
         1.5420e+00,  3.9328e+00, -3.8039e-01,  1.9230e+00,  2.9331e+00,
         4.1564e+00,  4.3931e+00, -1.4703e+00,  4.4385e+00,  2.7277e+00,
        -2.1532e+00, -1.0000e+09, -1.0000e+09, -1.0000e+09, -1.0000e+09,
        -1.0000e+09, -1.0000e+09, -1.0000e+09, -1.0000e+09, -1.0000e+09,
        -1.0000e+09, -1.0000e+09, -1.0000e+09, -1.0000e+09, -1.0000e+09],
       device='cuda:0', grad_fn=<SelectBackward0>) 

Attention M:  tensor([1.0000e+00, 2.1063e-27, 1.2979e-27, 5.7577e-28, 4.2475e-30, 9.4966e-42,
        1.0372e-40, 1.3887e-42, 1.3901e-41, 3.8170e-41, 1.2971e-40, 1.6436e-40,
        4.6663e-43, 1.7198e-40, 3.1084e-41, 2.3542e-43, 0.0

In [15]:
class MultiHeadAttention(nn.Module):
    def __init__(self):
        super(MultiHeadAttention, self).__init__()
        self.W_Q = nn.Linear(d_model, d_k * n_heads)
        self.W_K = nn.Linear(d_model, d_k * n_heads)
        self.W_V = nn.Linear(d_model, d_v * n_heads)
    def forward(self, Q, K, V, attn_mask):
        Q = Q.to(device)
        K = K.to(device)
        V = V.to(device)
        attn_mask = attn_mask.to(device)
        
        residual, batch_size = Q, Q.size(0)
        q_s = self.W_Q(Q).view(batch_size, -1, n_heads, d_k).transpose(1,2)
        k_s = self.W_K(K).view(batch_size, -1, n_heads, d_k).transpose(1,2)
        v_s = self.W_V(V).view(batch_size, -1, n_heads, d_v).transpose(1,2)

        attn_mask = attn_mask.unsqueeze(1).repeat(1, n_heads, 1, 1)
        
        _, context, attn = ScaledDotProductAttention().to(device)(q_s, k_s, v_s, attn_mask)
        attn = attn.to(device)
        context = context.transpose(1, 2).contiguous().view(batch_size, -1, n_heads * d_v)
        context = context.to(device)
        output = nn.Linear(n_heads * d_v, d_model).to(device)(context)
        return nn.LayerNorm(d_model).to(device)(output + residual), attn


In [16]:
emb = Embedding()
emb = emb.to(device)
input_ids = input_ids.to(device)
segment_ids = segment_ids.to(device)
embeds = emb(input_ids, segment_ids)
embeds = embeds.to(device)

attenM = get_attn_pad_mask(input_ids, input_ids)
attenM = attenM.to(device)

MHA= MultiHeadAttention().to(device)(embeds, embeds, embeds, attenM)

Output, A = MHA
Output = Output.to(device)
A = A.to(device)

A[0][0]

tensor([[0.1214, 0.0588, 0.0622, 0.0802, 0.0831, 0.0847, 0.0457, 0.0400, 0.0670,
         0.0251, 0.0418, 0.0611, 0.0507, 0.0462, 0.0683, 0.0635, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000],
        [0.0552, 0.0406, 0.0814, 0.0401, 0.0832, 0.0817, 0.0447, 0.0471, 0.0634,
         0.0530, 0.0430, 0.0687, 0.0550, 0.0560, 0.0821, 0.1047, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000],
        [0.0681, 0.0619, 0.0691, 0.0490, 0.0458, 0.0575, 0.0445, 0.0497, 0.0885,
         0.0600, 0.0760, 0.0728, 0.0604, 0.0533, 0.0936, 0.0497, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000, 0.0000,
         0.0000, 0.0000, 0.0000],
        [0.0624, 0.0639, 0.0718, 0.0417, 0.1110, 0.0535, 0.0582, 0.0400, 0.0611,
         0.0672, 0.0537, 0.0634, 0.0429, 0.0460, 0.0701, 0.0932, 0.0000, 0.0000,
       

In [17]:
class PoswiseFeedForwardNet(nn.Module):
    def __init__(self):
        super(PoswiseFeedForwardNet, self).__init__()
        self.fc1 = nn.Linear(d_model, d_ff)
        self.fc2 = nn.Linear(d_ff, d_model)

    def forward(self, x):
        return self.fc2(gelu(self.fc1(x)))


In [18]:
class EncoderLayer(nn.Module):
    def __init__(self):
        super(EncoderLayer, self).__init__()
        self.enc_self_attn = MultiHeadAttention()
        self.pos_ffn = PoswiseFeedForwardNet()

    def forward(self, enc_inputs, enc_self_attn_mask):
        enc_outputs, attn = self.enc_self_attn(enc_inputs, enc_inputs, enc_inputs, enc_self_attn_mask)
        attn = attn.to(device)
        enc_outputs = self.pos_ffn(enc_outputs)
        enc_outputs = enc_outputs.to(device)
        return enc_outputs, attn

In [19]:
class BERT(nn.Module):
    def __init__(self):
        super(BERT, self).__init__()
        self.embedding = Embedding()
        self.embedding = self.embedding.to(device)
        self.layers = nn.ModuleList([EncoderLayer() for _ in range(n_layers)])
        self.fc = nn.Linear(d_model, d_model)
        self.activ1 = nn.Tanh()
        self.linear = nn.Linear(d_model, d_model)
        self.activ2 = gelu
        self.norm = nn.LayerNorm(d_model)
        self.classifier = nn.Linear(d_model, 2)
        embed_weight = self.embedding.tok_embed.weight
        embed_weight.to(device)
        n_vocab, n_dim = embed_weight.size()
        self.decoder = nn.Linear(n_dim, n_vocab, bias=False)
        self.decoder.weight = embed_weight
        self.decoder_bias = nn.Parameter(torch.zeros(n_vocab))

    def forward(self, input_ids, segment_ids, masked_pos):
        input_ids = input_ids.to(device)
        segment_ids = segment_ids.to(device)
        masked_pos = masked_pos.to(device)

        output = self.embedding(input_ids, segment_ids)
        enc_self_attn_mask = get_attn_pad_mask(input_ids, input_ids).to(device)
        for layer in self.layers:
            output, enc_self_attn = layer(output, enc_self_attn_mask)
        h_pooled = self.activ1(self.fc(output[:, 0])) 
        logits_clsf = self.classifier(h_pooled) 

        masked_pos = masked_pos[:, :, None].expand(-1, -1, output.size(-1)) 
        
        h_masked = torch.gather(output, 1, masked_pos) 
        h_masked = self.norm(self.activ2(self.linear(h_masked)))
        logits_lm = self.decoder(h_masked) + self.decoder_bias 

        return logits_lm, logits_clsf

In [20]:
model = BERT()

model.to(device)

criterion = nn.CrossEntropyLoss()
optimizer = optim.Adam(model.parameters(), lr=0.001)

batch = make_batch()
input_ids, segment_ids, masked_tokens, masked_pos, isNext = map(torch.LongTensor, zip(*batch))
input_ids = input_ids.to(device)
segment_ids = segment_ids.to(device)
masked_tokens = masked_tokens.to(device)
masked_pos = masked_pos.to(device)
isNext = isNext.to(device)

for epoch in range(10):
    optimizer.zero_grad()
    logits_lm, logits_clsf = model(input_ids.to(device), segment_ids.to(device), masked_pos.to(device))
    loss_lm = criterion(logits_lm.transpose(1, 2), masked_tokens) 
    loss_lm = (loss_lm.float()).mean()
    loss_clsf = criterion(logits_clsf, isNext) 
    loss = loss_lm + loss_clsf
    if (epoch + 1) % 10 == 0:
        print('Epoch:', '%04d' % (epoch + 1), 'cost =', '{:.6f}'.format(loss))
    loss.backward()
    optimizer.step()

Epoch: 0010 cost = 60.073040


In [21]:
input_ids, segment_ids, masked_tokens, masked_pos, isNext = map(torch.LongTensor, zip(batch[0]))
print(text)
print([number_dict[w.item()] for w in input_ids[0] if number_dict[w.item()] != '[PAD]'])

input_ids = input_ids.to(device)
segment_ids = segment_ids.to(device)
masked_tokens = masked_tokens.to(device)
masked_pos = masked_pos.to(device)
isNext = isNext.to(device)

logits_lm, logits_clsf = model(input_ids, segment_ids, masked_pos)
logits_lm = logits_lm.data.max(2)[1][0].data.cpu().numpy()
print('masked tokens list : ',[pos.item() for pos in masked_tokens[0] if pos.item() != 0])
print('predict masked tokens list : ',[pos for pos in logits_lm if pos != 0])

logits_clsf = logits_clsf.data.max(1)[1].data.cpu().numpy()[0]
print('isNext : ', True if isNext else False)
print('predict isNext : ',True if logits_clsf else False)

Hello, how are you? I am Romeo.
Hello, Romeo My name is Juliet. Nice to meet you.
Nice meet you too. How are you today?
Great. My baseball team won the competition.
Oh Congratulations, Juliet
Thank you Romeo
['[CLS]', 'nice', 'meet', 'you', 'too', 'how', 'are', 'you', '[MASK]', '[SEP]', 'thank', 'you', '[MASK]', '[SEP]']
masked tokens list :  [5, 28]
predict masked tokens list :  [19, 19, 19, 19, 19]
isNext :  False
predict isNext :  False
