In [1]:
from datasets import load_dataset
dataset = load_dataset("yelp_review_full")

  from .autonotebook import tqdm as notebook_tqdm
Downloading readme: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 6.72k/6.72k [00:00<00:00, 6.73MB/s]
Downloading data: 100%|██████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 299M/299M [00:16<00:00, 18.0MB/s]
Downloading data: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 23.5M/23.5M [00:01<00:00, 17.7MB/s]
Generating train split: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 650000/650000 [00:01<00:00, 404164.07 examples/s]
Generating test split: 100%|██████████████████████████████████████████████████████████████████████

In [2]:
dataset.shape, dataset["train"][100]

({'train': (650000, 2), 'test': (50000, 2)},
 {'label': 0,
  'text': 'My expectations for McDonalds are t rarely high. But for one to still fail so spectacularly...that takes something special!\\nThe cashier took my friends\'s order, then promptly ignored me. I had to force myself in front of a cashier who opened his register to wait on the person BEHIND me. I waited over five minutes for a gigantic order that included precisely one kid\'s meal. After watching two people who ordered after me be handed their food, I asked where mine was. The manager started yelling at the cashiers for \\"serving off their orders\\" when they didn\'t have their food. But neither cashier was anywhere near those controls, and the manager was the one serving food to customers and clearing the boards.\\nThe manager was rude when giving me my order. She didn\'t make sure that I had everything ON MY RECEIPT, and never even had the decency to apologize that I felt I was getting poor service.\\nI\'ve eaten at va

In [30]:
dataset

DatasetDict({
    train: Dataset({
        features: ['label', 'text'],
        num_rows: 650000
    })
    test: Dataset({
        features: ['label', 'text'],
        num_rows: 50000
    })
})

In [3]:
from transformers import AutoTokenizer

tokenizer = AutoTokenizer.from_pretrained("google-bert/bert-base-cased")

To support symlinks on Windows, you either need to activate Developer Mode or to run Python as an administrator. In order to see activate developer mode, see this article: https://docs.microsoft.com/en-us/windows/apps/get-started/enable-your-device-for-development


In [4]:
def tokenize_function(examples):
    return tokenizer(examples["text"], padding="max_length", truncation=True)
    
tokenized_datasets = dataset.map(tokenize_function, batched=True)

Map: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 650000/650000 [05:12<00:00, 2081.35 examples/s]
Map: 100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 50000/50000 [00:24<00:00, 2045.80 examples/s]


In [5]:
tokenized_datasets.shape, tokenized_datasets["train"][100]

({'train': (650000, 5), 'test': (50000, 5)},
 {'label': 0,
  'text': 'My expectations for McDonalds are t rarely high. But for one to still fail so spectacularly...that takes something special!\\nThe cashier took my friends\'s order, then promptly ignored me. I had to force myself in front of a cashier who opened his register to wait on the person BEHIND me. I waited over five minutes for a gigantic order that included precisely one kid\'s meal. After watching two people who ordered after me be handed their food, I asked where mine was. The manager started yelling at the cashiers for \\"serving off their orders\\" when they didn\'t have their food. But neither cashier was anywhere near those controls, and the manager was the one serving food to customers and clearing the boards.\\nThe manager was rude when giving me my order. She didn\'t make sure that I had everything ON MY RECEIPT, and never even had the decency to apologize that I felt I was getting poor service.\\nI\'ve eaten at va

In [9]:
tokenized_datasets["train"][100].keys(), len(tokenized_datasets["train"][100]["input_ids"])

(dict_keys(['label', 'text', 'input_ids', 'token_type_ids', 'attention_mask']),
 512)

In [10]:
small_train_dataset = tokenized_datasets["train"].shuffle(seed=42).select(range(1000))
small_eval_dataset = tokenized_datasets["test"].shuffle(seed=42).select(range(1000))

In [11]:
small_train_dataset.shape, small_eval_dataset.shape

((1000, 5), (1000, 5))

# Load Pre_trained Model

In [12]:
from transformers import AutoModelForSequenceClassification

model = AutoModelForSequenceClassification.from_pretrained("google-bert/bert-base-cased", num_labels=5)

Some weights of BertForSequenceClassification were not initialized from the model checkpoint at google-bert/bert-base-cased and are newly initialized: ['classifier.bias', 'classifier.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


In [None]:
model

# Test before Fine-Tuning

In [24]:
test_sentence = " I ususally do not enjoy McDonalds."
res = tokenizer.encode(test_sentence)
res

[101, 146, 1366, 25034, 1193, 1202, 1136, 5548, 9092, 1116, 119, 102]

In [23]:
test_sentence = " I ususally do not enjoy McDonalds."
model_inputs = tokenizer([test_sentence], return_tensors="pt")
model_inputs

{'input_ids': tensor([[  101,   146,  1366, 25034,  1193,  1202,  1136,  5548,  9092,  1116,
           119,   102]]), 'token_type_ids': tensor([[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]]), 'attention_mask': tensor([[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]])}

In [27]:
output_logits = model(**model_inputs)
output_logits

SequenceClassifierOutput(loss=None, logits=tensor([[ 0.6461,  0.0360, -0.2371,  0.6171,  0.4696]],
       grad_fn=<AddmmBackward0>), hidden_states=None, attentions=None)

In [28]:
predictions = np.argmax(output_logits, axis=-1)
predictions

0

# Fine-Tuning

In [14]:
from transformers import TrainingArguments

training_args = TrainingArguments(output_dir="test_trainer")

In [17]:
import numpy as np
import evaluate

metric = evaluate.load("accuracy")
metric

EvaluationModule(name: "accuracy", module_type: "metric", features: {'predictions': Value(dtype='int32', id=None), 'references': Value(dtype='int32', id=None)}, usage: """
Args:
    predictions (`list` of `int`): Predicted labels.
    references (`list` of `int`): Ground truth labels.
    normalize (`boolean`): If set to False, returns the number of correctly classified samples. Otherwise, returns the fraction of correctly classified samples. Defaults to True.
    sample_weight (`list` of `float`): Sample weights Defaults to None.

Returns:
    accuracy (`float` or `int`): Accuracy score. Minimum possible value is 0. Maximum possible value is 1.0, or the number of examples input, if `normalize` is set to `True`.. A higher score means higher accuracy.

Examples:

    Example 1-A simple example
        >>> accuracy_metric = evaluate.load("accuracy")
        >>> results = accuracy_metric.compute(references=[0, 1, 2, 0, 1, 2], predictions=[0, 1, 1, 2, 1, 0])
        >>> print(results)
    

In [18]:
def compute_metrics(eval_pred):
    logits, labels = eval_pred
    predictions = np.argmax(logits, axis=-1)
    return metric.compute(predictions=predictions, references=labels)

In [19]:
from transformers import TrainingArguments, Trainer

training_args = TrainingArguments(output_dir="test_trainer", evaluation_strategy="epoch")

In [20]:
trainer = Trainer(
    model=model,
    args=training_args,
    train_dataset=small_train_dataset,
    eval_dataset=small_eval_dataset,
    compute_metrics=compute_metrics,
)

dataloader_config = DataLoaderConfiguration(dispatch_batches=None, split_batches=False, even_batches=True, use_seedable_sampler=True)


In [29]:
trainer.train()

Epoch,Training Loss,Validation Loss


KeyboardInterrupt: 

# Test after Fine-Tuning

In [None]:
new_tokenizer = AutoTokenizer.from_pretrained("~~")
new_model = AutoModelForSequenceClassification.from_pretrained("~~", num_labels=5)

In [None]:
test_sentence = " I ususally do not enjoy McDonalds."
res = tokenizer.encode(test_sentence)
res

In [None]:
test_sentence = " I ususally do not enjoy McDonalds."
model_inputs = tokenizer([test_sentence], return_tensors="pt")
model_inputs

In [None]:
output_logits = model(**model_inputs)
output_logits

In [None]:
predictions = np.argmax(output_logits, axis=-1)
predictions