# Task -> Evaluating the models on out sample data 

# Import Libraries 

In [1]:
from transformers import (
    AutoModelForCausalLM,
    AutoTokenizer,
    pipeline
)

import torch
from datasets import load_dataset

import gc
import re
import random

import nltk
from nltk.translate.bleu_score import sentence_bleu
from nltk.tokenize import word_tokenize
from rouge_score import rouge_scorer
from bert_score import score
nltk.download('punkt')

[nltk_data] Downloading package punkt to /home/pragyan/nltk_data...
[nltk_data]   Package punkt is already up-to-date!


True

# Get the models from their paths

In [2]:
# loading the fine tuned models 
base_model = "NousResearch/Llama-2-7b-chat-hf"
fine_tuned_model_oneInstruction = "llama2-finetunedSentimentClassificationOneInstruction"
fine_tuned_model_twoInstruction = "llama2-finetunedSentimentClassificationTwoInstruction"

# Generating some manual out of sample data for testing the base and fine tuned models

In [3]:
instructions = [
    "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:",
    
    "### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:",
    
    "Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:",
    
    "### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:",
    
    "### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:",
    
    "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:",
    
    " Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:",
    
    "### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:",
    
    "### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response:",
    
    "### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response:"
]


responsesManual = [
    "Artificial Intelligence transforming industries becoming a pivotal technology", 
    
    "¡Buen día! ¿Cómo estás?",
    
    "positive", 
    
    "1889", 
    
    "Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you",
    
    "tone is exciting", 
    
    "ruled the world", 
    
    "She felt tired and left the party early",
    
    "In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a Frost Dragon. In a fierce battle within Vorkahn's fortress, King struck down the sorcerer, restoring light to Liora. Celebrated as a hero, he retired peacefully, his courage forever remembered.",
    
    "Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on sunny days."]

In [4]:
# creating dictionary with prompt and response

out_of_sample_data = []

for i in range(len(instructions)):
    out_of_sample_data.append({"CompleteText": instructions[i] + " " + responsesManual[i],"Prompt": instructions[i], "Response": responsesManual[i]})

In [5]:
print(out_of_sample_data)

[{'CompleteText': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response: Artificial Intelligence transforming industries becoming a pivotal technology", 'Prompt': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:", 'Response': 'Artificial Intelligence transforming industries becoming a pivotal technology'}, {'CompleteText': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: ¡Buen día! ¿Cómo estás?', 'Prompt': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:

In [6]:
# store all the instruct here 
training_original = []
training_instances = []
training_instance_forPredictionList = []

# loop to create the instruction training data for fine tuning 
for j in out_of_sample_data:
    training_instance_forPredictionList.append(j['Prompt'])
    training_instances.append(j['Response'])
    training_original.append(j)

In [7]:
print(training_original[0])
print()
print(training_instance_forPredictionList[0])
print("---")
print(training_instances[0])

{'CompleteText': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response: Artificial Intelligence transforming industries becoming a pivotal technology", 'Prompt': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:", 'Response': 'Artificial Intelligence transforming industries becoming a pivotal technology'}

### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:
---
Artifici

# Generating the responses 

In [8]:
device = 'cuda' if torch.cuda.is_available() else 'cpu'
print(f'Using device: {device}')

Using device: cuda


In [9]:
def extract_candidate_sentiment(text):
    match = re.search(r"### Response:\s*(.*(?:\n.*)*)", text)
    return match.group(1) if match else None

## Using the Base Model 

In [10]:
# Load the tokenizer and model
tokenizer_base = AutoTokenizer.from_pretrained(base_model)
# model_oneInstruction = AutoModelForCausalLM.from_pretrained(fine_tuned_model_oneInstruction).to(device)
model_base = AutoModelForCausalLM.from_pretrained(base_model, device_map="auto")

totalBlueScore_base = []
totalROUEScore_base = []
totalBERTScore_base = []


for i in range(len(training_instance_forPredictionList)):
    pipe1 = pipeline(task="text-generation", model=model_base, tokenizer=tokenizer_base)
    result1 = pipe1(training_instance_forPredictionList[i], max_new_tokens=128)
    
    # reference -> ground truth 
    # candidate -> prediction 
    reference = str(training_original[i]['Response'])
    candidate_notPulled = result1[0]['generated_text']
    candiate_sentimentPulled = str(extract_candidate_sentiment(candidate_notPulled))
    print("Original: ", training_original[i])
    print("Prompt: ", training_instance_forPredictionList[i])
    print("Reference: ", reference)
    print("Candidate: ", candidate_notPulled)
    print("Candidate Pulled: ", candiate_sentimentPulled)
    print("**********")
    
    # reference_tokens = word_tokenize(reference)
    # candidate_tokens = word_tokenize(candiate_sentimentPulled)
    
    # print("Score: ")
    # Calculating BLEU score
    bleu_score = sentence_bleu([reference], candiate_sentimentPulled)
    # print(f"BLEU score: {bleu_score}")
    totalBlueScore_base.append(bleu_score)
    
    # Creating a ROUGE scorer
    scorer = rouge_scorer.RougeScorer(['rougeL'], use_stemmer=True)

    # Calculating ROUGE-L
    scores = scorer.score(reference, candiate_sentimentPulled)
    # print(f"ROUGE-L score: {scores['rougeL'].fmeasure}")
    totalROUEScore_base.append(scores['rougeL'].fmeasure)
    
    # Calculating BERTScore
    P, R, F1 = score([candiate_sentimentPulled], [reference], lang="en", verbose=True)

    # Printing the F1 score
    # print(f"BERTScore F1: {F1.mean().item()}")
    totalBERTScore_base.append(F1.mean().item())
    
    # print("-------")

Loading checkpoint shards:   0%|          | 0/2 [00:00<?, ?it/s]



Original:  {'CompleteText': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response: Artificial Intelligence transforming industries becoming a pivotal technology", 'Prompt': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:", 'Response': 'Artificial Intelligence transforming industries becoming a pivotal technology'}
Prompt:  ### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Res

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.08 seconds, 12.28 sentences/sec
Original:  {'CompleteText': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: ¡Buen día! ¿Cómo estás?', 'Prompt': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:', 'Response': '¡Buen día! ¿Cómo estás?'}
Prompt:  ### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:
Reference:  ¡Buen día! ¿Cómo estás?
Candidate:  ### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: ¡Buenos días! ¿Cómo estás? (Good morning! How are you?)
Candidate Pulled:  ¡Buenos días! ¿Cómo estás? (Good morning! How are you?)
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.04 seconds, 23.69 sentences/sec
Original:  {'CompleteText': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response: positive', 'Prompt': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:', 'Response': 'positive'}
Prompt:  Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:
Reference:  positive
Candidate:  Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response: Positive
Candidate Pulled:  Positive
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.47 sentences/sec
Original:  {'CompleteText': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response: 1889', 'Prompt': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:', 'Response': '1889'}
Prompt:  ### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:
Reference:  1889
Candidate:  ### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was con

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.83 sentences/sec
Original:  {'CompleteText': "### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response: Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you", 'Prompt': '### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:', 'Response': "Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you"}
Prompt:  ### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:
Reference:  Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you
Candidate:  ### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response: Bob: I've been good, just busy with work and stuff. 

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 30.78 sentences/sec
Original:  {'CompleteText': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response: tone is exciting", 'Prompt': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:", 'Response': 'tone is exciting'}
Prompt:  ### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:
Reference:  tone is exciting
Candidate:  ### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I bel

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.01 sentences/sec
Original:  {'CompleteText': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response: ruled the world', 'Prompt': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:', 'Response': 'ruled the world'}
Prompt:   Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:
Reference:  ruled the world
Candidate:   Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response: lived in a magnificent castle with his parents, the king and queen.

Please provide the correct answer for the blank space.
Candidate Pulled:  lived in a magnificent castle with his parents, the king and queen.

Please provide the correct answer for the blank space.
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.69 sentences/sec
Original:  {'CompleteText': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response: She felt tired and left the party early', 'Prompt': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:', 'Response': 'She felt tired and left the party early'}
Prompt:  ### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:
Reference:  She felt tired and left the party early
Candidate:  ### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response: She chose to depart from the gathering before its conclusion because she experienced fatigue.
Candidate Pulled:  She chose to depart from the gathering before its conclusion because she experien

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.64 sentences/sec
Original:  {'CompleteText': "### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response: In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a Frost Dragon. In a fierce battle within Vorkahn's fortress, King struck down the sorcerer, restoring light to Liora. Celebrated as a hero, he retired peacefully, his courage forever remembered.", 'Prompt': '### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response:', 'Response': "In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a F

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 29.71 sentences/sec
Original:  {'CompleteText': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response: Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on sunny days.', 'Prompt': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response:', 'Response': 'Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on s

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 30.62 sentences/sec


In [11]:
del tokenizer_base
del model_base
del pipe1

gc.collect()

11292

## Using fine tuned Model with only one instruction -> custom

In [12]:
# Load the tokenizer and model
tokenizer_oneInstruction = AutoTokenizer.from_pretrained(fine_tuned_model_oneInstruction)
# model_oneInstruction = AutoModelForCausalLM.from_pretrained(fine_tuned_model_oneInstruction).to(device)
model_oneInstruction = AutoModelForCausalLM.from_pretrained(fine_tuned_model_oneInstruction, device_map="auto")

totalBlueScore_oneInstruction = []
totalROUEScore_oneInstruction = []
totalBERTScore_oneInstruction = []


for i in range(len(training_instance_forPredictionList)):
    pipe1 = pipeline(task="text-generation", model=model_oneInstruction, tokenizer=tokenizer_oneInstruction)
    result1 = pipe1(training_instance_forPredictionList[i], max_new_tokens=128)
    
    # reference -> ground truth 
    # candidate -> prediction 
    reference = str(training_original[i]['Response'])
    candidate_notPulled = result1[0]['generated_text']
    candiate_sentimentPulled = str(extract_candidate_sentiment(candidate_notPulled))
    print("Original: ", training_original[i])
    print("Prompt: ", training_instance_forPredictionList[i])
    print("Reference: ", reference)
    print("Candidate: ", candidate_notPulled)
    print("Candidate Pulled: ", candiate_sentimentPulled)
    print("**********")
    
    
    # reference_tokens = word_tokenize(reference)
    # candidate_tokens = word_tokenize(candiate_sentimentPulled)
    
    # print("Score: ")
    # Calculating BLEU score
    bleu_score = sentence_bleu([reference], candiate_sentimentPulled)
    # print(f"BLEU score: {bleu_score}")
    totalBlueScore_oneInstruction.append(bleu_score)
    
    # Creating a ROUGE scorer
    scorer = rouge_scorer.RougeScorer(['rougeL'], use_stemmer=True)

    # Calculating ROUGE-L
    scores = scorer.score(reference, candiate_sentimentPulled)
    # print(f"ROUGE-L score: {scores['rougeL'].fmeasure}")
    totalROUEScore_oneInstruction.append(scores['rougeL'].fmeasure)
    
    # Calculating BERTScore
    P, R, F1 = score([candiate_sentimentPulled], [reference], lang="en", verbose=True)

    # Printing the F1 score
    # print(f"BERTScore F1: {F1.mean().item()}")
    totalBERTScore_oneInstruction.append(F1.mean().item())
    
    # print("-------")

Loading checkpoint shards:   0%|          | 0/2 [00:00<?, ?it/s]



Original:  {'CompleteText': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response: Artificial Intelligence transforming industries becoming a pivotal technology", 'Prompt': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:", 'Response': 'Artificial Intelligence transforming industries becoming a pivotal technology'}
Prompt:  ### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Res

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.27 sentences/sec
Original:  {'CompleteText': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: ¡Buen día! ¿Cómo estás?', 'Prompt': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:', 'Response': '¡Buen día! ¿Cómo estás?'}
Prompt:  ### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:
Reference:  ¡Buen día! ¿Cómo estás?
Candidate:  ### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: Good morning! I'm doing well, thanks for asking.  I'm just getting ready to go to work. ### Sentiment: Neutral ### Confidence: High ### Topic: Work ### Emotion: Neutral ### Strength: 7 ### Weakness: 3 ### Word: Good ### Sentiment: Neutral ### Confidence: High ### Topic: Work ### Emotion: Neutral ### Strength: 7 ### Weakness: 3 ### Word: Good ### Sentiment:

The hypothesis contains 0 counts of 3-gram overlaps.
Therefore the BLEU score evaluates to 0, independently of
how many N-gram overlaps of lower order it contains.
Consider using lower n-gram order or use SmoothingFunction()
The hypothesis contains 0 counts of 4-gram overlaps.
Therefore the BLEU score evaluates to 0, independently of
how many N-gram overlaps of lower order it contains.
Consider using lower n-gram order or use SmoothingFunction()
Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.74 sentences/sec
Original:  {'CompleteText': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response: positive', 'Prompt': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:', 'Response': 'positive'}
Prompt:  Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:
Reference:  positive
Candidate:  Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response: @michelle_mcdonald I'm glad you enjoyed it! ### Sentiment: Positive ### Hashtags: #movies #fantasticbeasts ### Sentiment: Positive ### Hashtags: #movies #fantasticbeasts ### Sentiment: Positive ### Hashtags: #movies #fantasticbeast

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.82 sentences/sec
Original:  {'CompleteText': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response: 1889', 'Prompt': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:', 'Response': '1889'}
Prompt:  ### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:
Reference:  1889
Candidate:  ### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was con

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.87 sentences/sec
Original:  {'CompleteText': "### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response: Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you", 'Prompt': '### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:', 'Response': "Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you"}
Prompt:  ### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:
Reference:  Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you
Candidate:  ### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response: Bob: I've been good, just busy with work.  I'm going

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.47 sentences/sec
Original:  {'CompleteText': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response: tone is exciting", 'Prompt': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:", 'Response': 'tone is exciting'}
Prompt:  ### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:
Reference:  tone is exciting
Candidate:  ### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I bel

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.55 sentences/sec
Original:  {'CompleteText': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response: ruled the world', 'Prompt': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:', 'Response': 'ruled the world'}
Prompt:   Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:
Reference:  ruled the world
Candidate:   Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response: was very unhappy because he had no one to love him.

Please select one of the following options: ### Options: 1. He was very unhappy because he had no one to love him. ### Options: 2. He was very unhappy because he had no one to love him. ### Options: 3. He was very unhappy because he had no one to love him. ### Options:

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.79 sentences/sec
Original:  {'CompleteText': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response: She felt tired and left the party early', 'Prompt': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:', 'Response': 'She felt tired and left the party early'}
Prompt:  ### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:
Reference:  She felt tired and left the party early
Candidate:  ### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response: She decided to leave the party early because she felt exhausted. ### Instruction: Paraphrase the following sentence: ### Input: He is going to the store to buy some milk. ### Response: He is goi

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.39 sentences/sec
Original:  {'CompleteText': "### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response: In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a Frost Dragon. In a fierce battle within Vorkahn's fortress, King struck down the sorcerer, restoring light to Liora. Celebrated as a hero, he retired peacefully, his courage forever remembered.", 'Prompt': '### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response:', 'Response': "In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a F

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 30.34 sentences/sec
Original:  {'CompleteText': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response: Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on sunny days.', 'Prompt': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response:', 'Response': 'Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on s

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 30.98 sentences/sec


In [13]:
del tokenizer_oneInstruction
del model_oneInstruction
del pipe1

gc.collect()

26388

## Using fine tuned Model with two instruction -> Custom + Alpaca

In [14]:
# Load the tokenizer and model
tokenizer_twoInstruction = AutoTokenizer.from_pretrained(fine_tuned_model_twoInstruction)
model_twoInstruction = AutoModelForCausalLM.from_pretrained(fine_tuned_model_twoInstruction, device_map="auto")

totalBlueScore_twoInstruction = []
totalROUEScore_twoInstruction = []
totalBERTScore_twoInstruction = []


for i in range(len(training_instance_forPredictionList)):
    pipe1 = pipeline(task="text-generation", model=model_twoInstruction, tokenizer=tokenizer_twoInstruction)
    result1 = pipe1(training_instance_forPredictionList[i], max_new_tokens=1)
    
    # reference -> ground truth 
    # candidate -> prediction 
    reference = str(training_original[i]['Response'])
    candidate_notPulled = result1[0]['generated_text']
    candiate_sentimentPulled = str(extract_candidate_sentiment(candidate_notPulled))
    print("Original: ", training_original[i])
    print("Prompt: ", training_instance_forPredictionList[i])
    print("Reference: ", reference)
    print("Candidate: ", candiate_sentimentPulled)
    print("**********")
    
    
    # reference_tokens = word_tokenize(reference)
    # candidate_tokens = word_tokenize(candiate_sentimentPulled)
    
    # print("Score: ")
    # Calculating BLEU score
    bleu_score = sentence_bleu([reference], candiate_sentimentPulled)
    # print(f"BLEU score: {bleu_score}")
    totalBlueScore_twoInstruction.append(bleu_score)
    
    # Creating a ROUGE scorer
    scorer = rouge_scorer.RougeScorer(['rougeL'], use_stemmer=True)

    # Calculating ROUGE-L
    scores = scorer.score(reference, candiate_sentimentPulled)
    # print(f"ROUGE-L score: {scores['rougeL'].fmeasure}")
    totalROUEScore_twoInstruction.append(scores['rougeL'].fmeasure)
    
    # Calculating BERTScore
    P, R, F1 = score([candiate_sentimentPulled], [reference], lang="en", verbose=True)

    # Printing the F1 score
    # print(f"BERTScore F1: {F1.mean().item()}")
    totalBERTScore_twoInstruction.append(F1.mean().item())
    
    # print("-------")

Loading checkpoint shards:   0%|          | 0/2 [00:00<?, ?it/s]



Original:  {'CompleteText': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response: Artificial Intelligence transforming industries becoming a pivotal technology", 'Prompt': "### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Response:", 'Response': 'Artificial Intelligence transforming industries becoming a pivotal technology'}
Prompt:  ### Instruction: Extract the main points from the following text: ### Input: Artificial Intelligence is transforming various industries, improving efficiency, and enabling new possibilities. It's becoming a pivotal technology in the modern world. ### Res

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.98 sentences/sec
Original:  {'CompleteText': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response: ¡Buen día! ¿Cómo estás?', 'Prompt': '### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:', 'Response': '¡Buen día! ¿Cómo estás?'}
Prompt:  ### Task: Translate the following sentence into Spanish: ### Input: Good morning! How are you? ### Response:
Reference:  ¡Buen día! ¿Cómo estás?
Candidate:  ¡
**********


The hypothesis contains 0 counts of 2-gram overlaps.
Therefore the BLEU score evaluates to 0, independently of
how many N-gram overlaps of lower order it contains.
Consider using lower n-gram order or use SmoothingFunction()
Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.73 sentences/sec
Original:  {'CompleteText': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response: positive', 'Prompt': 'Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:', 'Response': 'positive'}
Prompt:  Your task is to classify following text as positive, negative, or neutral. ### Text: The movie was fantastic! I loved every moment of it. ### Response:
Reference:  positive
Candidate:  Pos
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 33.23 sentences/sec
Original:  {'CompleteText': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response: 1889', 'Prompt': '### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:', 'Response': '1889'}
Prompt:  ### Instruction: Answer the question based on the given passage: ### Input: The Eiffel Tower is one of the most famous landmarks in Paris. It was constructed in 1889. ### Question: When was the Eiffel Tower constructed? ### Response:
Reference:  1889
Candidate:  The
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.69 sentences/sec
Original:  {'CompleteText': "### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response: Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you", 'Prompt': '### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:', 'Response': "Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you"}
Prompt:  ### To Do: Generate a conversation between two characters, Alice and Bob: ### Input: Alice: Hi Bob, how have you been? ### Response:
Reference:  Alice: Hi Bob, how have you been? Bob: I am good. How are you? Alice: I'm good too. Thank you
Candidate:  Bob
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.54 sentences/sec
Original:  {'CompleteText': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response: tone is exciting", 'Prompt': "### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:", 'Response': 'tone is exciting'}
Prompt:  ### Instruction: Analyze the tone of the following text: ### Input: I'm really excited about the new project we're starting. It's going to be challenging, but I believe in our team's ability to succeed. ### Response:
Reference:  tone is exciting
Candidate:  The
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.80 sentences/sec
Original:  {'CompleteText': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response: ruled the world', 'Prompt': ' Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:', 'Response': 'ruled the world'}
Prompt:   Complete the following text: ### Input: Once upon a time, in a faraway kingdom, there was a young prince who ### Response:
Reference:  ruled the world
Candidate:  was
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.90 sentences/sec
Original:  {'CompleteText': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response: She felt tired and left the party early', 'Prompt': '### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:', 'Response': 'She felt tired and left the party early'}
Prompt:  ### Instruction: Paraphrase the following sentence: ### Input: She decided to leave the party early because she felt tired. ### Response:
Reference:  She felt tired and left the party early
Candidate:  She
**********


Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 32.76 sentences/sec
Original:  {'CompleteText': "### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response: In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a Frost Dragon. In a fierce battle within Vorkahn's fortress, King struck down the sorcerer, restoring light to Liora. Celebrated as a hero, he retired peacefully, his courage forever remembered.", 'Prompt': '### Instruction: Write a short story about a brave knight on a quest to save a kingdom: ### Input: A brave knight named King ### Response:', 'Response': "In the Kingdom of Liora, the brave knight King set out to defeat the dark sorcerer Vorkahn, who had cast a shadow over the realm. He fought through treacherous forests and mountains, overcoming shadow wolves and a F

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.68 sentences/sec
Original:  {'CompleteText': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response: Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on sunny days.', 'Prompt': '### Instruction: Generate five similar examples based on the following text: ### Input: John loves hiking in the mountains every weekend. ### Response:', 'Response': 'Sarah enjoys biking along the coastal trails every morning., Emily takes her kayak out to the lake for a paddle each afternoon., David finds solace in running through the forest paths every evening., Lisa spends her Saturdays exploring the desert dunes., Tom goes fishing by the riverbank on s

Some weights of RobertaModel were not initialized from the model checkpoint at roberta-large and are newly initialized: ['roberta.pooler.dense.bias', 'roberta.pooler.dense.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


calculating scores...
computing bert embedding.


  0%|          | 0/1 [00:00<?, ?it/s]

computing greedy matching.


  0%|          | 0/1 [00:00<?, ?it/s]

done in 0.03 seconds, 31.75 sentences/sec




In [15]:
del tokenizer_twoInstruction
del model_twoInstruction
del pipe1

gc.collect()

26388

# Final Score - Base / One Instruction / Two Instruction Model

In [16]:
print("Final Scores: ")

print("------------  Base -----------------------------")
meanBleuBase= sum(totalBlueScore_base) / len(totalBlueScore_base)
meanROUEScoreBase = sum(totalROUEScore_base) / len(totalROUEScore_base)
meanBERTScoreBase = sum(totalBERTScore_base) / len(totalBERTScore_base)
print("Blue: ", meanBleuBase)
print("ROUGE: ", meanROUEScoreBase)
print("BERTScore: ", meanBERTScoreBase)

print("------------  One Instruction -----------------------------")
meanBleuOneInstruction = sum(totalBlueScore_oneInstruction) / len(totalBlueScore_oneInstruction)
meanROUEScoreOneInstruction = sum(totalROUEScore_oneInstruction) / len(totalROUEScore_oneInstruction)
meanBERTScoreOneInstruction = sum(totalBERTScore_oneInstruction) / len(totalBERTScore_oneInstruction)
print("Blue: ", meanBleuOneInstruction)
print("ROUGE: ", meanROUEScoreOneInstruction)
print("BERTScore: ", meanBERTScoreOneInstruction)


print("------------  Two Instruction -----------------------------")
meanBleuTwoInstruction = sum(totalBlueScore_twoInstruction) / len(totalBlueScore_twoInstruction)
meanROUEScoreTwoInstruction = sum(totalROUEScore_twoInstruction) / len(totalROUEScore_twoInstruction)
meanBERTScoreTwoInstruction = sum(totalBERTScore_twoInstruction) / len(totalBERTScore_twoInstruction)
print("Blue: ", meanBleuTwoInstruction)
print("ROUGE: ", meanROUEScoreTwoInstruction)
print("BERTScore: ", meanBERTScoreTwoInstruction)

Final Scores: 
------------  Base -----------------------------
Blue:  0.2587568509400362
ROUGE:  0.2820792978871937
BERTScore:  0.8771790385246276
------------  One Instruction -----------------------------
Blue:  0.09717143739586283
ROUGE:  0.08688978390412586
BERTScore:  0.8119335114955902
------------  Two Instruction -----------------------------
Blue:  1.757239464727628e-45
ROUGE:  0.034254307783719545
BERTScore:  0.7778677463531494
