In [2]:
import pandas as pd
import json
import matplotlib.pyplot as plt

from nltk.corpus import stopwords
from nltk.stem import SnowballStemmer
import re

import requests
import json
from tqdm import tqdm
from emoji import demojize
from nltk.tokenize import TweetTokenizer
import os
from datetime import datetime
import sys
import os
sys.path.append("../src")
import prompt_utils

import numpy as np

import pandas as pd
from collections import Counter

# Define a list of filenames to load
filenames = ["../data/labeled_data/generic_test_0.json"]

df_train = pd.DataFrame()
df_test = pd.DataFrame()
df_valid = pd.DataFrame()

# Load all JSON data and concatenate into one DataFrame
for filename in filenames:
    with open(filename) as f:
        data = json.load(f)
    df_train = pd.DataFrame(data["train"])
    df_test = pd.DataFrame(data["test"])
    df_valid = pd.DataFrame(data["valid"])

def normalizeToken(token):
    lowercased_token = token.lower()
    if token.startswith("@"):
        return "@USER"
    elif lowercased_token.startswith("http") or lowercased_token.startswith("www"):
        return "[url]"
    elif len(token) == 1:
        return demojize(token)
    else:
        if token == "’":
            return "'"
        elif token == "…":
            return "..."
        else:
            return token
    
def normalizeTweet(tweet):
    tokens = TweetTokenizer().tokenize(tweet.replace("’", "'"))
    normTweet = " ".join([normalizeToken(token) for token in tokens])

    normTweet = (
        normTweet.replace("n 't", "n't")
            .replace("ca n't", "can't")
            .replace("ai n't", "ain't")
    )
    normTweet = (
        normTweet.replace("p . m .", "pm")
            .replace("p . m", "pm")
            .replace("a . m .", "am")
            .replace("a . m", "am")
    )
    return " ".join(normTweet.split())

def api(prompt):
    import requests

# For local streaming, the websockets are hosted without ssl - http://
HOST = 'http://127.0.0.1:5000'
URI = f'{HOST}/api/v1/generate'

# For reverse-proxied streaming, the remote will likely host with ssl - https://
# URI = 'https://your-uri-here.trycloudflare.com/api/v1/generate'

def get_response(request_params, prompt, context):
    request_params['prompt'] = prompt
    request_params['context'] = context

    response = requests.post(URI, json=request_params)

    if response.status_code == 200:
        result = response.json()['results'][0]['text']
        #print(prompt + result)
        return result
    else:
    	print(response)

def get_base_request_params(max_new_tokens = 200, stopping_strings = []):
    return {
        'prompt': None,
        'context': None,
        'max_new_tokens': 200,
        'do_sample': True,
        'temperature': 0.7,
        'top_p': 0.1,
        'typical_p': 1,
        'repetition_penalty': 1.2,
        'encoder_repetition_penalty': 1.0,
        'top_k': 40,
        'min_length': 0,
        'no_repeat_ngram_size': 0,
        'num_beams': 1,
        'penalty_alpha': 0,
        'length_penalty': 1,
        'early_stopping': False,
        'seed': -1,
        #'add_bos_token': True,
        #'truncation_length': 2048,
        #'ban_eos_token': False,
        #'skip_special_tokens': True,
        'stopping_strings': stopping_strings
    }

def get_vicuna_multi_label_no_fine_tune_v01_old(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format [\"topic 1\", \"topic 2\", ...].\n\nTopics:\n[\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\"]"
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\nTopics in format [\"topic 1\", \"topic 2\", ...]: \n[\""
    return prompt, ""

def get_vicuna_multi_label_no_fine_tune_v01(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion.\n\nTopics:\n\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\""
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\n[\""
    
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["\n", "Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

def get_vicuna_multi_label_no_fine_tune_v02(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format [\"a\", \"b\", ...].\n\nTopics:\n\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\""
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant: "
    
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["\n", "Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

def get_vicuna_multi_label_few_shot_prompt_only_classification_n_random_example(tweet_text, example_tweets):
    example_tweets_str = ""
    for example_tweet in example_tweets:
        replaced_topic = str(example_tweet[1]).replace('\'', '"')
        example_tweets_str += f"\nTweet: {prompt_utils.normalize_tweet_simplified(example_tweet[0])}\nTopics: {replaced_topic}\n"

    prompt = f"### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion.\n\n\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\"\n{example_tweets_str}\nTweet: {tweet_text}\n###Assistant:"
    request_params = get_base_request_params()
    request_params["max_new_tokens"] = 400
    request_params["stopping_strings"] = ["\n", "### Human:", "Human:", "###"]
    return prompt, "", request_params

def get_vicuna_multi_label_no_fine_tune_explanation_first_v01(tweet_text):
    instruction = "### Human:\nOut of the following list of topics, choose topics that best fit the text in your opinion. Start your response with \"Explanation:\" and end it with the list of chosen topics \"Topics:\".\n\nTopics:\n\"War/Terror\", \"Conspiracy Theory\", \"Education\", \"Election Campaign\", \"Environment\", \"Government/Public\", \"Health\", \"Immigration/Integration\", \"Justice/Crime\", \"Labor/Employment\", \"Macroeconomics/Economic Regulation\", \"Media/Journalism\", \"Religion\", \"Science/Technology\""
    prompt = f"{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\nExplanation: \n"
    request_params = get_base_request_params()
    request_params["max_new_tokens"] = 400
    request_params["stopping_strings"] = ["### Human:", "Human:", "###"]
    return prompt, "", request_params

def get_vicuna_multi_label_v01_no_alpaca_format(tweet_text):
    instruction = "Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format [\"a\", \"b\", ...].\n\nTopics:\n['War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology']"
    prompt = f"### Human:\n{instruction}\n\nTweet: {tweet_text}\n\n### Assistant:\n"

    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Human:", "Assistant:"]
    return prompt, "", request_params

def get_vicuna_multi_label_v01(tweet_text):
    instruction = "Assign multilabel labels to the following tweet. Choose out of the following list of labels where 'Others' is only assigned if no other label fits: ['War/Terror', 'Conspiracy Theory', Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology', 'Others']"
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

#TODO: retrain with "Others", rules arent formatted correctly
def get_vicuna_multi_label_v02(tweet_text):
    instruction = "Out of the following list of topics, choose topics that best fit the text in your opinion. Here are the rules:\n- Use clear indicators for labeling.\n- If unclear, output 'Others'. \n- No speculation.\n- Don't mix content from different operations.\n- Refer to Oxford definitions for topics.\n\nTopics: 'War/Terror', 'Conspiracy Theory', 'Education', 'Election Campaign', 'Environment', 'Government/Public', 'Health', 'Immigration/Integration', 'Justice/Crime', 'Labor/Employment', 'Macroeconomics/Economic Regulation', 'Media/Journalism', 'Religion', 'Science/Technology'"
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 200
    return prompt, "", request_params

def get_vicuna_binary_v01(label, tweet_text):
    instruction = f"Classify the input based on if it's about {label}. Use 1 or 0 as output."
    prompt = f"Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.\n\n### Instruction:\n{instruction}\n\n### Input:\n{tweet_text}\n\n### Response:\n"
    
    request_params = get_base_request_params()
    request_params["stopping_strings"] = ["Instruction:", "###"]
    request_params["max_new_tokens"] = 30
    return prompt, "", request_params

def get_random_examples(df, label, exclude_tweet, n, random_state = 42):
    """_summary_

    Args:
        df (pd.dataframe): df to retrieve random examples
        label (str): label to check random examples with, if set to "multilabel", it will return the true label in string format of the random tweets
        exclude_tweet (str): text of tweet to be excluded from returned list
        n (int): amount of random tweets retrieved

    Returns:
        list(tuple): returns a list of tuples of format (tweet_text, 1/0 depending on whether it pertains to the label) 
    """
    # Exclude the specific tweet
    df = df[df['text'] != exclude_tweet]
    
    # Sample n random examples
    sampled_df = df.sample(n=n, random_state=random_state)

    values = sampled_df['text'].values
    if label == "multilabel":
        labels = sampled_df['annotations'].values
    else:
        labels = sampled_df['annotations'].apply(lambda x: int(label in x)).values
    
    # Return a list of tuples, each containing the tweet text and its annotations
    return list(zip(values, labels))

In [9]:
get_vicuna_multi_label_few_shot_prompt_only_classification_n_random_example("awdawd", get_random_examples(df_train, "multilabel", "awdawdawd", 3))

('### Human:\nOut of the following list of topics, choose topics that best fit the tweet in your opinion.\n\n"War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"\n\nTweet: A 53 - year-old Zambian national Willie Oliver Ndembela who served as Special Assistant to the African Union Commissioner of Trade and Industry of His Excellency Albert Muchanga has passed away in Addis Ababa , Ethiopia . [url]\nTopics: ["Others"]\n\nTweet: “ Black Lives Matter ” Issues Revolting Anti-Christian Message on Christmas [url] [url]\nTopics: ["Government/Public", "Religion"]\n\nTweet: RT @USER : Remind them of the #ArmyCommander who never finished the #LRA rebellion but instead fled the war front . [url]\nTopics: ["Government/Public", "War/Terror"]\n\nTweet: awdawd\n###Assistant:',
 '',
 {'pro

In [10]:
str(get_random_examples(df_train, "multilabel", "awdawdawd", 3)[0][1])

"['Others']"

In [14]:
models_to_test_names = ["binary_others_v01"]
model_funcs = [get_vicuna_binary_v01]
label = "Others"
dataframes = [df_test]
lora = True
dataframes_names = ["test"]

for i, df in enumerate(dataframes):

    for model_name, model_func in zip(models_to_test_names, model_funcs):
        print("Starting with model: " + model_name)
        print("----------------------------------")
        df_tmp = df.copy()
        df_tmp["prompt"] = ""
        df_tmp["context"] = ""
        new_column_name = "response"
        if lora:
            model_name = "lora/" + model_name
        output_folder = f"../data/vicuna_4bit/{model_name}/"
        if not os.path.exists(output_folder):
            os.makedirs(output_folder)

        for idx, row in tqdm(df_tmp.iterrows(), total=df_tmp.shape[0]):

            tweet_text = normalizeTweet(row["text"])
            df_tmp.loc[lambda df: df['id'] == row["id"], 'normalized_tweet'] = tweet_text

            if label != "":
                prompt, context, request_params = model_func(label, tweet_text)
            else:
                prompt, context, request_params = model_func(tweet_text)
            df_tmp.at[idx, "prompt"] = prompt
            df_tmp.at[idx, "context"] = context

            response = get_response(request_params, prompt, "")

            # Save the response in the 'api_results' column
            df_tmp.loc[lambda df: df['id'] == row["id"], new_column_name] = response
            if (idx + 1) % 100 == 0:
                output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
                df_tmp.to_csv(output_path, index=False)
                print(f"Saved progress at index {idx}")
                print("Sample Prompt: ", prompt)
                print("Sample Annotation: ", response)

        output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
        df_tmp.to_csv(output_path, index=False)        
            # Save the request_params as a JSON file in the output folder
        with open(os.path.join(output_folder, 'request_params.json'), 'w') as f:
            json.dump(request_params, f, indent=4)

Starting with model: binary_others_v01
----------------------------------


 10%|█         | 100/1000 [00:49<06:45,  2.22it/s]

Saved progress at index 99
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
RT @USER : When you fight all your battles on your kness ( prayer ) , you will always win . #MenGather2020

### Response:

Sample Annotation:  1




 20%|██        | 200/1000 [01:28<04:39,  2.86it/s]

Saved progress at index 199
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
RT @USER : #TweetRequest | What music videos would you like to watch on MTV Base ? Tweet us NOW with the hashtag & name of vid :front-facing_baby_chick: htt …

### Response:

Sample Annotation:  1




 30%|███       | 300/1000 [02:13<05:05,  2.29it/s]

Saved progress at index 299
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
However , the #Syrian troops repelled the offensive and retained the occupied lines . #Militants suffered losses and retreated .

### Response:

Sample Annotation:  0




 40%|████      | 400/1000 [02:51<02:46,  3.61it/s]

Saved progress at index 399
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
Militants of terrorist groups constantly arrange provocations by shelling the positions of Syrian government forces in western Aleppo . [url]

### Response:

Sample Annotation:  0




 50%|█████     | 500/1000 [03:35<03:41,  2.26it/s]

Saved progress at index 499
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
RT @USER : A working person in Iraq is a working person is Texas is a working person in Berlin ; war is the desire to kill this tr …

### Response:

Sample Annotation:  0




 60%|██████    | 600/1000 [04:15<01:50,  3.61it/s]

Saved progress at index 599
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
The #US is completing the training of one more terrorist group at #AlTanf military base . #Syria heading for the next series of terrorist attacks very soon [url]

### Response:

Sample Annotation:  0




 70%|███████   | 700/1000 [04:57<02:10,  2.29it/s]

Saved progress at index 699
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
RT @USER : #SleepunderthenetUg Door to door distribution of mosquito nets in Buwambwa sub county , Namisindwa District . kick malaria …

### Response:

Sample Annotation:  1




 80%|████████  | 800/1000 [05:37<00:55,  3.63it/s]

Saved progress at index 799
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
RT @USER =: 🇺🇸 #US military is being deployed to the #Greek Section ! Op-ed by @USER [url] #Turkey #Russia # U …

### Response:

Sample Annotation:  0




 90%|█████████ | 900/1000 [06:19<00:44,  2.25it/s]

Saved progress at index 899
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
@USER Smashing it up Tomorrow Again @USER @USER #PremiereWednesday THE DAWN Kasanga [url]

### Response:

Sample Annotation:  1




100%|██████████| 1000/1000 [07:00<00:00,  2.38it/s]

Saved progress at index 999
Sample Prompt:  Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.

### Instruction:
Classify the input based on if it's about Others. Use 1 or 0 as output.

### Input:
VIDEO : #Women'sMarch Linda Sarsour Told Maddow “ Islamic Children Are Executed In The U . S . ” [url] [url]

### Response:

Sample Annotation:  0







In [13]:
models_to_test_names = ["multilabel_no_fine_tune_v01_3"]
model_funcs = [get_vicuna_multi_label_no_fine_tune_v01_3]
label = ""
dataframes = [df_test]
lora = False
dataframes_names = ["test"]

for i, df in enumerate(dataframes):

    for model_name, model_func in zip(models_to_test_names, model_funcs):
        print("Starting with model: " + model_name)
        print("----------------------------------")
        df_tmp = df.copy()
        df_tmp["prompt"] = ""
        df_tmp["context"] = ""
        new_column_name = "response"
        if lora:
            model_name = "lora/" + model_name
        output_folder = f"../data/vicuna_4bit/{model_name}/"
        if not os.path.exists(output_folder):
            os.makedirs(output_folder)

        for idx, row in tqdm(df_tmp.iterrows(), total=df_tmp.shape[0]):

            tweet_text = normalizeTweet(row["text"])
            df_tmp.loc[lambda df: df['id'] == row["id"], 'normalized_tweet'] = tweet_text

            if label != "":
                prompt, context, request_params = model_func(label, tweet_text)
            else:
                prompt, context, request_params = model_func(tweet_text)
            df_tmp.at[idx, "prompt"] = prompt
            df_tmp.at[idx, "context"] = context

            response = get_response(request_params, prompt, "")

            # Save the response in the 'api_results' column
            df_tmp.loc[lambda df: df['id'] == row["id"], new_column_name] = response
            if (idx + 1) % 100 == 0:
                output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
                df_tmp.to_csv(output_path, index=False)
                print(f"Saved progress at index {idx}")
                print("Sample Prompt: ", prompt)
                print("Sample Annotation: ", response)

        output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
        df_tmp.to_csv(output_path, index=False)        
            # Save the request_params as a JSON file in the output folder
        with open(os.path.join(output_folder, 'request_params.json'), 'w') as f:
            json.dump(request_params, f, indent=4)

Starting with model: multilabel_no_fine_tune_v01_3
----------------------------------


 10%|█         | 100/1000 [02:08<19:56,  1.33s/it]

Saved progress at index 99
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: RT @USER : When you fight all your battles on your kness ( prayer ) , you will always win . #MenGather2020

### Assistant: 
Sample Annotation:  🙏🏽❤️‍💪 Out of respect for our diverse audience and to adhere to ethical guidelines, please refrain from promoting religious content or political agendas. Let's focus on uniting people through positive messages & everyday life experiences! #SocialMediaPeace pic.twitter.com/Yuq3V89QlP



 20%|██        | 200/1000 [04:14<16:47,  1.26s/it]

Saved progress at index 199
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: RT @USER : #TweetRequest | What music videos would you like to watch on MTV Base ? Tweet us NOW with the hashtag & name of vid :front-facing_baby_chick: htt …

### Assistant: 
Sample Annotation:  🎶 Music Videos 📺👀 Which ones do YOU want to see?! 🤔 Use #MTVBaseVideoFest & tag @user to request your favs! Let's make this a party! 🎉💥 #MTVBase #MusicVideos #PartyTime

❌ NOTE: This response is not an accurate reflection of the original prompt and should be disregarded. Please ignore if you came across it by accident. If you are ex

 30%|███       | 300/1000 [06:23<08:28,  1.38it/s]

Saved progress at index 299
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: However , the #Syrian troops repelled the offensive and retained the occupied lines . #Militants suffered losses and retreated .

### Assistant: 
Sample Annotation:  ​



 40%|████      | 400/1000 [08:22<06:45,  1.48it/s]

Saved progress at index 399
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: Militants of terrorist groups constantly arrange provocations by shelling the positions of Syrian government forces in western Aleppo . [url]

### Assistant: 
Sample Annotation:  ​



 50%|█████     | 500/1000 [10:43<15:27,  1.85s/it]

Saved progress at index 499
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: RT @USER : A working person in Iraq is a working person is Texas is a working person in Berlin ; war is the desire to kill this tr …

### Assistant: 
Sample Annotation:  ‣ War/Terror
‣ Justice/Crime
‣ Government/Public
‣ Labor/Employment
‣ Macroeconomics/Economic Regulation
‣ Media/Journalism
‣ Immigration/Integration
‣ Science/Technology
‣ Environment
‣ Health
‣ Education
‣ Election Campaign
‣ Religion
‣ Conspiracy Theory



 60%|██████    | 600/1000 [12:44<07:17,  1.09s/it]

Saved progress at index 599
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: The #US is completing the training of one more terrorist group at #AlTanf military base . #Syria heading for the next series of terrorist attacks very soon [url]

### Assistant: 
Sample Annotation:  ​



 70%|███████   | 700/1000 [14:44<01:59,  2.50it/s]

Saved progress at index 699
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: RT @USER : #SleepunderthenetUg Door to door distribution of mosquito nets in Buwambwa sub county , Namisindwa District . kick malaria …

### Assistant: 
Sample Annotation:  ​



 80%|████████  | 800/1000 [16:54<02:53,  1.15it/s]

Saved progress at index 799
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: RT @USER =: 🇺🇸 #US military is being deployed to the #Greek Section ! Op-ed by @USER [url] #Turkey #Russia # U …

### Assistant: 
Sample Annotation:  ​



 90%|█████████ | 900/1000 [18:28<01:44,  1.04s/it]

Saved progress at index 899
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: @USER Smashing it up Tomorrow Again @USER @USER #PremiereWednesday THE DAWN Kasanga [url]

### Assistant: 
Sample Annotation:  ​



100%|██████████| 1000/1000 [20:10<00:00,  1.21s/it]

Saved progress at index 999
Sample Prompt:  ### Human:
Out of the following list of topics, choose topics that best fit the tweet in your opinion. Output in format ["a", "b", ...].

Topics:
["War/Terror", "Conspiracy Theory", "Education", "Election Campaign", "Environment", "Government/Public", "Health", "Immigration/Integration", "Justice/Crime", "Labor/Employment", "Macroeconomics/Economic Regulation", "Media/Journalism", "Religion", "Science/Technology"]

Tweet: VIDEO : #Women'sMarch Linda Sarsour Told Maddow “ Islamic Children Are Executed In The U . S . ” [url] [url]

### Assistant: 
Sample Annotation:  ​






In [15]:
# code to count the number of rows in df_tmp where "0" is in the column test_vicuna_binary_war_v01
df_tmp[df_tmp["testvicuna_binary_war_v01"].str.contains("0")].shape[0]

0

In [7]:
if not os.path.exists(output_folder):
    os.makedirs(output_folder)
output_path = os.path.join(output_folder, f'{dataframes_names[i]}_generic_test_0.csv')
df_tmp.to_csv(output_path, index=False)        
    # Save the request_params as a JSON file in the output folder
with open(os.path.join(output_folder, 'request_params.json'), 'w') as f:
    json.dump(request_params, f, indent=4)