# Current Method Analysis

### Import libraries

In [None]:
import pandas as pd
import numpy as np
from scipy.stats import ttest_rel
from sklearn.model_selection import train_test_split
from tqdm import tqdm
import re
import nltk
from fuzzywuzzy import fuzz

### Load data

In [None]:
df = pd.read_csv('final_df.csv')

## Accuracy Current Method

In [None]:
def calculate_accuracy(df, col1, col2):
    # Ensure data is in the correct type and handle any case sensitivity issues
    df[col1] = df[col1].astype(str).str.lower()
    df[col2] = df[col2].astype(str).str.lower()

    correct_predictions = 0
    total_predictions = len(df)

    for a, b in zip(df[col1], df[col2]):
        if fuzz.partial_ratio(a, b) >= 80:  # Adjust the threshold as needed
            correct_predictions += 1

    accuracy = correct_predictions / total_predictions if total_predictions > 0 else 0
    return accuracy

In [None]:
accuracy = calculate_accuracy(df, 'Organization Name', 'True Organization')
print(f"Accuracy: {accuracy:.2%}")

## Compare document level accuracy of current method with NLP models

In [None]:
train, test = train_test_split(df, test_size=0.3, random_state=42)

In [None]:
test['accuracy'] = np.where(test['Organization Name'] == test['True Organization'], 1, 0)
current_array = test['accuracy'].to_numpy()
spacy_array = np.array([0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 1, 1, 1, 0, 0, 0, 1, 0, 1, 0, 0, 0, 1, 1, 1, 1, 1, 0, 1, 1, 1, 0, 1, 1, 0, 0, 1, 1, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 1, 0, 0, 1, 0, 0, 0, 0, 1, 0, 1, 0, 1, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 1, 0, 1, 0, 0, 0, 0, 0, 1, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 1, 0, 1, 1, 0, 0, 1, 0, 1, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 1, 0, 1, 0, 0, 1, 1, 0, 0, 1, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 1, 0, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 1, 0, 0, 1, 1, 1, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 0, 0, 1, 0, 0, 0, 1, 0, 1, 0, 0, 1, 0, 0, 1, 0, 0, 1, 0, 0, 0, 1, 0, 1, 1, 1, 0, 1, 1, 1, 0, 1, 1, 1, 0, 0, 1, 1, 0, 1, 0, 1, 0, 1, 1, 0, 1, 0, 1, 1, 0, 0, 0, 1, 1])
gpt_array = np.array([1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 0, 1, 1, 1, 1, 0, 0, 1, 1, 0, 1, 0, 0, 1, 0, 0, 1, 1, 1, 0, 1, 1, 1, 0, 1, 0, 1, 1, 1, 1, 1, 1, 0, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 0, 1, 0, 1, 0, 1, 1, 1, 1, 0, 1, 1, 0, 1, 1, 0, 0, 1, 1, 0, 1, 1, 1, 1, 0, 1, 0, 1, 0, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 0, 1, 0, 0, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 0, 1, 0, 1, 0, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 0, 1, 1, 0, 1, 1, 1, 0, 0, 1, 0, 0, 1, 1, 0, 0, 1, 1, 0, 0, 1, 1, 1, 1])
robbert_array = np.array([1, 1, 1, 0, 0, 1, 1, 0, 0, 1, 0, 0, 0, 1, 0, 1, 0, 1, 1, 1, 1, 0, 0, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 1, 0, 1, 1, 0, 1, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 1, 1, 0, 0, 0, 1, 1, 0, 1, 1, 1, 0, 1, 1, 0, 1, 0, 0, 1, 0, 1, 0, 0, 0, 0, 1, 1, 1, 0, 1, 1, 1, 1, 0, 1, 0, 0, 1, 0, 0, 0, 1, 1, 1, 0, 1, 1, 0, 0, 0, 0, 0, 1, 0, 0, 1, 0, 1, 0, 0, 1, 0, 1, 0, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 0, 1, 0, 1, 0, 0, 0, 0, 1, 0, 0, 1, 0, 1, 0, 0, 1, 1, 1, 0, 1, 1, 1, 1, 0, 1, 0, 0, 1, 1, 0, 0, 1, 1, 1, 0, 0, 1, 0, 1, 0, 1, 1, 1, 1, 1, 0, 0, 1, 0, 0, 1, 0, 0, 1, 1, 1, 0, 0, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 0, 1, 0, 1, 0, 1, 0, 1, 1, 0, 1, 1, 0, 0, 1, 1, 1, 1, 0, 1, 0, 1, 0, 0, 0, 0, 1, 0, 1, 0, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0])
llama_array = np.array([])

### Paired t-tests to test significance

#### SpaCy

In [None]:
t_stat, p_value = ttest_rel(current_array, spacy_array)

print(f"T-statistic: {t_stat}")
print(f"P-value: {p_value}")

T-statistic: -11.802518164707418
P-value: 3.622598205899714e-26


#### GPT-3.5

In [None]:
# current & gpt
t_stat, p_value = ttest_rel(current_array, gpt_array)

print(f"T-statistic: {t_stat}")
print(f"P-value: {p_value}")

T-statistic: -31.571695073666238
P-value: 2.7489666923369945e-92


#### RobBERT

In [None]:
# current & robbert
t_stat, p_value = ttest_rel(current_array, robbert_array)

print(f"T-statistic: {t_stat}")
print(f"P-value: {p_value}")

T-statistic: -17.570885009014198
P-value: 1.6630807808172687e-46


#### Llama3

In [None]:
# current & llama
t_stat, p_value = ttest_rel(current_array, llama_array)

print(f"T-statistic: {t_stat}")
print(f"P-value: {p_value}")

ValueError: unequal length arrays

## Get actuals
Documents where the organization name is mentioned (1) or not (0) for precision and recall

In [None]:
def get_actuals(df):
    labeled_data = []

    for index, row in tqdm(df.iterrows(), total=len(df), desc="Creating training data"):
        text = row['Cleaned Text']
        org_name = row['True Organization']
        
        if pd.isnull(org_name):
            continue
        
        # Break down the organization name into words and escape special characters
        org_words = org_name.split()
        escaped_org_words = [re.escape(word) for word in org_words]
        
        # Pattern to find any of the organization name words
        pattern = rf"\b({'|'.join(escaped_org_words)})\b"
        
        entities = []
        
        # Find all matches in the text
        for match in re.finditer(pattern, text):
            start_index = match.start()
            end_index = match.end()
            
            # Add the entity to the list
            entities.append((start_index, end_index, 'ORG'))
        
        if len(entities) > 0:
            has_entity = 1
        else:
            has_entity = 0
        labeled_data.append((text, has_entity))
    return labeled_data

Creating training data:   0%|          | 0/269 [00:00<?, ?it/s]

Creating training data: 100%|██████████| 269/269 [00:00<00:00, 3936.36it/s]


In [None]:
actuals_test = get_actuals(test)
actuals = [1 if entry[1] else 0 for entry in actuals_test]

[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]
269 269 269 269
