In [1]:
import pandas as pd
import json
import matplotlib.pyplot as plt

from nltk.corpus import stopwords
from nltk.stem import SnowballStemmer
import re

import requests
import json
from tqdm import tqdm
from emoji import demojize
from nltk.tokenize import TweetTokenizer
import os
from datetime import datetime

import numpy as np

import pandas as pd
from collections import Counter

# Define a list of filenames to load
filenames = ["../data/labeled_data/generic_test_0.json"]

df_train = pd.DataFrame()
df_test = pd.DataFrame()
df_valid = pd.DataFrame()

# Load all JSON data and concatenate into one DataFrame
for filename in filenames:
    with open(filename) as f:
        data = json.load(f)
    df_train = pd.DataFrame(data["train"])
    df_test = pd.DataFrame(data["test"])
    df_valid = pd.DataFrame(data["valid"])

def normalizeToken(token):
    lowercased_token = token.lower()
    if token.startswith("@"):
        return "@USER"
    elif lowercased_token.startswith("http") or lowercased_token.startswith("www"):
        return "[url]"
    elif len(token) == 1:
        return demojize(token)
    else:
        if token == "’":
            return "'"
        elif token == "…":
            return "..."
        else:
            return token
    
def normalizeTweet(tweet):
    tokens = TweetTokenizer().tokenize(tweet.replace("’", "'"))
    normTweet = " ".join([normalizeToken(token) for token in tokens])

    normTweet = (
        normTweet.replace("n 't", "n't")
            .replace("ca n't", "can't")
            .replace("ai n't", "ain't")
    )
    normTweet = (
        normTweet.replace("p . m .", "pm")
            .replace("p . m", "pm")
            .replace("a . m .", "am")
            .replace("a . m", "am")
    )
    return " ".join(normTweet.split())

def api(prompt):
    import requests

# For local streaming, the websockets are hosted without ssl - http://
HOST = 'http://127.0.0.1:5000'
URI = f'{HOST}/api/v1/generate'

# For reverse-proxied streaming, the remote will likely host with ssl - https://
# URI = 'https://your-uri-here.trycloudflare.com/api/v1/generate'

def get_response(request_params, prompt, context):
    request_params['prompt'] = prompt
    request_params['context'] = context

    response = requests.post(URI, json=request_params)

    if response.status_code == 200:
        result = response.json()['results'][0]['text']
        #print(prompt + result)
        return result
    else:
    	print(response)

def get_base_request_params(max_new_tokens = 200, stopping_strings = []):
    return {
        'prompt': None,
        'context': None,
        'max_new_tokens': 200,
        'do_sample': True,
        'temperature': 0.7,
        'top_p': 0.1,
        'typical_p': 1,
        'repetition_penalty': 1.2,
        'encoder_repetition_penalty': 1.0,
        'top_k': 40,
        'min_length': 0,
        'no_repeat_ngram_size': 0,
        'num_beams': 1,
        'penalty_alpha': 0,
        'length_penalty': 1,
        'early_stopping': False,
        'seed': -1,
        #'add_bos_token': True,
        #'truncation_length': 2048,
        #'ban_eos_token': False,
        #'skip_special_tokens': True,
        'stopping_strings': stopping_strings
    }

def get_vicuna_multi_label_no_fine_tune_v01(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format [\"topic 1\", \"topic 2\", ...].\n\nTopics:\n[\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\"]"
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\nTopics in format [\"topic 1\", \"topic 2\", ...]: \n[\""
    return prompt, ""

def get_vicuna_multi_label_no_fine_tune_explanation_first_v01(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the text in your opinion. Start your response with \"Explanation:\" and end it with the list of chosen topics \"Topics:\".\n\nTopics:\n[\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\"]"
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\nExplanation: \n"
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["### Human:", "Human:", "###"]
    return prompt, "", request_params

def get_vicuna_multi_label_v01_retest(tweet_text):
    instruction = "Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ['topic 1', 'topic 2', ...].\n\nTopics:\n['War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology']"
    prompt = f"### Human:\n{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\n"

    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Human:", "Assistant:"]
    return prompt, "", request_params

def get_vicuna_multi_label_v01(tweet_text):
    instruction = "Assign multilabel labels to the following tweet. Choose out of the following list of labels where 'Others' is only assigned if no other label fits: ['War/Terror', 'Conspiracy Theory', Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology', 'Others']"
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

#TODO: retrain with "Others", rules arent formatted correctly
def get_vicuna_multi_label_v02(tweet_text):
    instruction = "Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:\n- Use clear indicators for labeling.\n- If unclear, output 'Others'. \n- No speculation.\n- Don't mix content from different operations.\n- Refer to Oxford definitions for topics.\n\nTopics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'"
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

def get_vicuna_binary_v01(label, tweet_text):
    instruction = f"Classify the input based on if it's about {label}. Use 1 (True) or 0 (False) as output."
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 30
    return prompt, "", request_params

In [2]:
get_vicuna_multi_label_v01("awd")

("Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\nAssign multilabel labels to the following tweet. Choose out of the following list of labels where 'Others' is only assigned if no other label fits: ['War/Terror', 'Conspiracy Theory', Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology', 'Others']\n\n### Input:\nawd\n\n### Response:\n",
 '',
 {'prompt': None,
  'context': None,
  'max_new_tokens': 200,
  'do_sample': True,
  'temperature': 0.7,
  'top_p': 0.1,
  'typical_p': 1,
  'repetition_penalty': 1.2,
  'encoder_repetition_penalty': 1.0,
  'top_k': 40,
  'min_length': 0,
  'no_repeat_ngram_size': 0,
  'num_beams': 1,
  'penalty_alpha': 0,
  'length_penalty': 1,
  'early_stopp

In [3]:
models_to_test_names = ["multilabel_with_rules_v02"]
model_funcs = [get_vicuna_multi_label_v02]
label = ""
dataframes = [df_test]
lora = True
dataframes_names = ["test"]

for i, df in enumerate(dataframes):

    for model_name, model_func in zip(models_to_test_names, model_funcs):
        print("Starting with model: " + model_name)
        print("----------------------------------")
        df_tmp = df.copy()
        df_tmp["prompt"] = ""
        df_tmp["context"] = ""
        new_column_name = "response"
        if lora:
            model_name = "lora/" + model_name
        output_folder = f"../data/vicuna_4bit/{model_name}/"
        if not os.path.exists(output_folder):
            os.makedirs(output_folder)

        for idx, row in tqdm(df_tmp.iterrows(), total=df_tmp.shape[0]):

            tweet_text = normalizeTweet(row["text"])
            df_tmp.loc[lambda df: df['id'] == row["id"], 'normalized_tweet'] = tweet_text

            if label != "":
                prompt, context, request_params = model_func(label, tweet_text)
            else:
                prompt, context, request_params = model_func(tweet_text)
            df_tmp.at[idx, "prompt"] = prompt
            df_tmp.at[idx, "context"] = context

            response = get_response(request_params, prompt, "")

            # Save the response in the 'api_results' column
            df_tmp.loc[lambda df: df['id'] == row["id"], new_column_name] = response
            if (idx + 1) % 100 == 0:
                output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
                df_tmp.to_csv(output_path, index=False)
                print(f"Saved progress at index {idx}")
                print("Sample Prompt: ", prompt)
                print("Sample Annotation: ", response)

        output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
        df_tmp.to_csv(output_path, index=False)        
            # Save the request_params as a JSON file in the output folder
        with open(os.path.join(output_folder, 'request_params.json'), 'w') as f:
            json.dump(request_params, f, indent=4)

Starting with model: multilabel_with_rules_v02
----------------------------------


  0%|          | 0/1000 [00:00<?, ?it/s]

 10%|█         | 100/1000 [02:08<18:54,  1.26s/it]

Saved progress at index 99
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
RT @USER : When you fight all your battles on your kness ( prayer ) , you will always win . #MenGather2020

### Response:

Sample Annotation:  ['Religion']




 20%|██        | 200/1000 [04:13<16:23,  1.23s/it]

Saved progress at index 199
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
RT @USER : #TweetRequest | What music videos would you like to watch on MTV Base ? Tweet us NOW with the hashtag & name of vid :front-facing_baby_chick: htt …

### Response:

Sample Annotation:  ['Other

 30%|███       | 300/1000 [06:22<14:43,  1.26s/it]

Saved progress at index 299
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
However , the #Syrian troops repelled the offensive and retained the occupied lines . #Militants suffered losses and retreated .

### Response:

Sample Annotation:  ['War/Terror']




 40%|████      | 400/1000 [08:31<13:27,  1.35s/it]

Saved progress at index 399
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
Militants of terrorist groups constantly arrange provocations by shelling the positions of Syrian government forces in western Aleppo . [url]

### Response:

Sample Annotation:  ['War/Terror']




 50%|█████     | 500/1000 [10:38<10:12,  1.22s/it]

Saved progress at index 499
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
RT @USER : A working person in Iraq is a working person is Texas is a working person in Berlin ; war is the desire to kill this tr …

### Response:

Sample Annotation:  ['War/Terror']




 60%|██████    | 600/1000 [12:41<08:56,  1.34s/it]

Saved progress at index 599
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
The #US is completing the training of one more terrorist group at #AlTanf military base . #Syria heading for the next series of terrorist attacks very soon [url]

### Response:

Sample Annotation:  ['Co

 70%|███████   | 700/1000 [14:48<06:07,  1.22s/it]

Saved progress at index 699
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
RT @USER : #SleepunderthenetUg Door to door distribution of mosquito nets in Buwambwa sub county , Namisindwa District . kick malaria …

### Response:

Sample Annotation:  ['Health']




 80%|████████  | 800/1000 [16:53<04:11,  1.26s/it]

Saved progress at index 799
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
RT @USER =: 🇺🇸 #US military is being deployed to the #Greek Section ! Op-ed by @USER [url] #Turkey #Russia # U …

### Response:

Sample Annotation:  ['War/Terror']




 90%|█████████ | 900/1000 [18:56<01:55,  1.16s/it]

Saved progress at index 899
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
@USER Smashing it up Tomorrow Again @USER @USER #PremiereWednesday THE DAWN Kasanga [url]

### Response:

Sample Annotation:  ['Others']




100%|██████████| 1000/1000 [21:04<00:00,  1.26s/it]

Saved progress at index 999
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:
- Use clear indicators for labeling.
- If unclear, output 'Others'. 
- No speculation.
- Don't mix content from different operations.
- Refer to Oxford definitions for topics.

Topics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'

### Input:
VIDEO : #Women'sMarch Linda Sarsour Told Maddow “ Islamic Children Are Executed In The U . S . ” [url] [url]

### Response:

Sample Annotation:  ['Conspiracy Theory', 'Immigration/Integration']







In [15]:
# code to count the number of rows in df_tmp where "0" is in the column test_vicuna_binary_war_v01
df_tmp[df_tmp["testvicuna_binary_war_v01"].str.contains("0")].shape[0]

0

In [5]:
prompt

"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\nClassify the input based on if it's about War/Terror. Use 1 (True) or 0 (False) as output.\n\n### Input:\n#Crimea : What to Expect from #Ukraine and the West [url] #russiainvadesukraine\n\n### Response:\n"

In [4]:
response

'1'

In [7]:
if not os.path.exists(output_folder):
    os.makedirs(output_folder)
output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
df_tmp.to_csv(output_path, index=False)        
    # Save the request_params as a JSON file in the output folder
with open(os.path.join(output_folder, 'request_params.json'), 'w') as f:
    json.dump(request_params, f, indent=4)