# Fix pathing

In [1]:
import sys


sys.path.append("../..")


In [2]:
import constants

import os


constants.PROJECT_DIRECTORY_PATH = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(constants.PROJECT_DIRECTORY_PATH))))


# Imports

In [3]:
import plotter
import datahandler

import matplotlib.pyplot
import numpy as np
import pandas as pd
import seaborn as sns
import IPython.display


# Constants

In [4]:
FOLDER_NAME = "2024_05_03_13_08_04_NONE"

FOLDER_PATH = os.path.join(os.path.dirname(constants.PROJECT_DIRECTORY_PATH), "Simulator", "data", FOLDER_NAME)


In [5]:
data_preprocessor = datahandler.DataPreprocessorOUS_V2()
data_preprocessor.execute()

data_loader = datahandler.DataLoader(datahandler.DataPreprocessorOUS_V2)
data_loader.execute(False, False, True)


Cleaning dataset: 100%|██████████| 2/2 [00:00<?, ?it/s]
Processing dataset: 100%|██████████| 2/2 [00:00<?, ?it/s]
Enhancing dataset: 100%|██████████| 2/2 [00:00<?, ?it/s]
Loading dataset: 100%|██████████| 2/2 [00:03<00:00,  1.52s/it]


# Methods

In [6]:
def load_csv(fileName):
    df = pd.read_csv(os.path.join(FOLDER_PATH, fileName + ".csv"))

    response_time_cols = [
        'duration_incident_creation',
        'duration_resource_appointment',
        'duration_resource_preparing_departure',
        'duration_dispatching_to_scene'
    ]
    df['total_response_time'] = df[response_time_cols].sum(axis=1)

    return df


In [7]:
def print_results(df: pd.DataFrame):
    # Define the criteria for response times
    criteria = {
        ('A', True): 12 * 60,
        ('A', False): 25 * 60,
        ('H', True): 30 * 60,
        ('H', False): 40 * 60
    }

    # Function to calculate compliance for each group
    def calculate_compliance(group, triage, urban):
        limit = criteria.get((triage, urban))
        if limit is not None:
            return (group['total_response_time'] < limit).mean()
        return None

    # Calculate statistics and compliance for each group
    results = []
    for (triage, urban), group in df.groupby(['triage_impression_during_call', 'urban']):
        mean = group['total_response_time'].mean()
        median = group['total_response_time'].median()
        compliance = calculate_compliance(group, triage, urban)
        results.append({
            'Triage': triage,
            'Urban': urban,
            'Mean (sec)': mean,
            'Median (sec)': median,
            'Compliance': compliance
        })

    stats = pd.DataFrame(results)

    # Convert mean and median to minutes
    stats['Mean (min)'] = (stats['Mean (sec)'] / 60)
    stats['Median (min)'] = (stats['Median (sec)'] / 60)
    stats.drop(columns=['Mean (sec)', 'Median (sec)'], inplace=True)

    # Map urban values to Yes/No
    stats['Urban'] = stats['Urban'].map({True: 'Yes', False: 'No'})
    
    # Sort values
    stats.sort_values(by=["Urban", "Triage"], ascending=[False, True], inplace=True)
    
    # Display the DataFrame
    formatted_stats = stats.style.format({
        'Mean (min)': "{:.2f}",
        'Median (min)': "{:.2f}",
        'Compliance': "{:.2%}"
    }).hide(axis='index')
    IPython.display.display(formatted_stats)


In [8]:
def boxplot_time_at_steps_modified(
    dataframe: pd.DataFrame,
    triage_impression: str = None
):
    title = "Time Taken At Each Step of the Incident"

    if triage_impression is not None:
        # Filter the dataframe without overwriting the original one
        temp_df = dataframe[dataframe["triage_impression_during_call"] == triage_impression].copy()
        title += f" ({triage_impression})"
    else:
        # Use the original dataframe if no triage_impression filter is applied
        temp_df = dataframe.copy()

    steps = {
        "Creating Incident": "duration_incident_creation",
        "Appointing Resource": "duration_resource_appointment",
        "Resource to Start Task": "duration_resource_preparing_departure",
        "Dispatching to Scene": "duration_dispatching_to_scene",
        "At Scene": "duration_at_scene",
        "Dispatching to Hospital": "duration_dispatching_to_hospital",
        "At Hospital": "duration_at_hospital"
    }

    # Calculating durations for each step
    plot_data = [(temp_df[duration_column][temp_df[duration_column] > 0] / 60) for step, duration_column in steps.items()]

    # Plotting
    matplotlib.pyplot.figure(figsize=(8, 4))
    matplotlib.pyplot.boxplot(plot_data[::-1], labels=list(steps.keys())[::-1], vert=False, patch_artist=True, showfliers=False)
    matplotlib.pyplot.title(title)
    matplotlib.pyplot.xlabel("Time in Minutes")
    matplotlib.pyplot.xticks()
    matplotlib.pyplot.show()


In [9]:
def boxplot_time_at_steps(
    historic_dataframe: pd.DataFrame,
    simulated_dataframe: pd.DataFrame,
    bounds: tuple[str, str] = None,
    triage_impressions: list[str] = ["A", "H", "V1"],
):
    historic_df = historic_dataframe.copy()
    simulated_df = simulated_dataframe.copy()
    
    # filter historic data by time frame
    if bounds is not None:
        start_bound, end_bound = pd.to_datetime(bounds[0]), pd.to_datetime(bounds[1])
        historic_df = historic_df[(historic_df['time_call_received'] >= start_bound) & (historic_df['time_call_received'] <= end_bound)]

    # calculate duration at each stage in minutes (simulated dataframe already has this calculated)
    historic_steps = {
        "duration_incident_creation": ("time_call_received", "time_incident_created"),
        "duration_resource_appointment": ("time_incident_created", "time_resource_appointed"),
        "duration_resource_preparing_departure": ("time_resource_appointed", "time_ambulance_dispatch_to_scene"),
        "duration_dispatching_to_scene": ("time_ambulance_dispatch_to_scene", "time_ambulance_arrived_at_scene"),
        "duration_at_scene": ("time_ambulance_arrived_at_scene", "time_ambulance_dispatch_to_hospital", "time_ambulance_available"),
        "duration_dispatching_to_hospital": ("time_ambulance_dispatch_to_hospital", "time_ambulance_arrived_at_hospital"),
        "duration_at_hospital": ("time_ambulance_arrived_at_hospital", "time_ambulance_available")
    }

    for step, times in historic_steps.items():
        if len(times) == 3:
            historic_df.loc[historic_df[times[1]].isna(), step] = (historic_df[times[2]] - historic_df[times[0]]).dt.total_seconds() / 60
            historic_df.loc[~historic_df[times[1]].isna(), step] = (historic_df[times[1]] - historic_df[times[0]]).dt.total_seconds() / 60
        else:
            historic_df[step] = (historic_df[times[1]] - historic_df[times[0]]).dt.total_seconds() / 60
    
    # convert seconds to minutes and replace zeros in simulated data to nan
    for step in historic_steps.keys():
        simulated_df[step] /= 60

    simulated_df[["duration_dispatching_to_hospital", "duration_at_hospital"]] = simulated_df[["duration_dispatching_to_hospital", "duration_at_hospital"]].replace(0, np.nan)

    # plot data
    matplotlib.pyplot.figure(figsize=(8, 8))

    data = []
    positions = []
    labels = []
    colors = []
    stripes = []

    i = 0
    for step in historic_steps.keys():
        labels.append(f"{step}")
        for triage in triage_impressions:
            for df, stripe, alpha in zip([historic_df, simulated_df], [False, True], [1.00, 0.65]):
                data.append(df[df['triage_impression_during_call'] == triage][step].dropna())

                labels.append("")

                if (triage == "A"):
                    colors.append([1.00, 0.37, 0.28, alpha])
                elif (triage == "H"):
                    colors.append([0.12, 0.56, 1.00, alpha])
                else:
                    colors.append([0.20, 0.80, 0.20, alpha])

                stripes.append(stripe)

                positions.append(i)
                i += 0.75
        labels.pop()
        i += 1

    bplot = matplotlib.pyplot.boxplot(
        data[::-1],
        labels=labels[::-1],
        positions=positions,
        vert=False,
        patch_artist=True,
        showfliers=True
    )

    for patch, color, stripe in zip(bplot["boxes"], colors[::-1], stripes[::-1]):
        patch.set_facecolor(color)

    matplotlib.pyplot.title("Time Taken At Each Step of the Incident")
    matplotlib.pyplot.xlabel("Time in Minutes")
    matplotlib.pyplot.xticks()
    matplotlib.pyplot.show()


In [10]:
def table_results(urban = True, strategies = ["closest", "random"]):
    def calculate_compliance(df):
        if df is None:
            return None

        # Criteria for both triage types in both urban and rural settings
        criteria = {
            ('A', True): 12 * 60,  # 12 minutes for triage 'A' in urban areas
            ('A', False): 25 * 60, # 25 minutes for triage 'A' in rural areas
            ('H', True): 30 * 60,  # 30 minutes for triage 'H' in urban areas
            ('H', False): 40 * 60  # 40 minutes for triage 'H' in rural areas
        }

        # Initialize counters
        total_compliant_cases = 0
        total_cases = 0
        
        # Check necessary columns exist
        if 'triage_impression_during_call' in df.columns and 'urban' in df.columns and 'total_response_time' in df.columns:
            for triage in ['A', 'H']:
                filtered_df = df[(df['triage_impression_during_call'] == triage) & (df['urban'] == urban)]
                limit = criteria.get((triage, urban))
                if not filtered_df.empty:
                    # Count compliant cases for this triage type
                    compliant_cases = filtered_df['total_response_time'] < limit
                    total_compliant_cases += compliant_cases.sum()
                    total_cases += len(filtered_df)
            if total_cases > 0:
                # Calculate overall compliance rate across both triage types
                overall_compliance = total_compliant_cases / total_cases
                return overall_compliance
            else:
                return None  # Return None if there are no cases to evaluate
        else:
            # If required columns are missing, return None to indicate an issue with the data format.
            return None

    # Settings combinations
    prioritize_triages = [False, True]
    response_restricteds = [False, True]
    schedule_breaks = [False, True]

    results = []

    for strategy in strategies:
        for prioritize_triage in prioritize_triages:
            for response_restricted in response_restricteds:
                for schedule_break in schedule_breaks:
                    filename = f"events_strategy={strategy}_prioritizeTriage={'true' if prioritize_triage else 'false'}_responseRestricted={'true' if response_restricted else 'false'}_scheduleBreaks={'true' if schedule_break else 'false'}"
                    df = load_csv(filename)
                    compliance = calculate_compliance(df)
                    if compliance is not None:
                        results.append({
                            "Strategy": strategy,
                            "Prioritize Triage": prioritize_triage,
                            "Response Restricted": response_restricted,
                            "Schedule Breaks": schedule_break,
                            "Compliance": compliance
                        })

    # Create DataFrame
    results_df = pd.DataFrame(results)
    # Pivot Table for visualization
    pivot_table = results_df.pivot_table(index=["Schedule Breaks", "Prioritize Triage", "Response Restricted"],
                                        columns=["Strategy"], 
                                        values="Compliance")
    # Color coding from green to red
    cm = sns.light_palette("green", as_cmap=True, n_colors=8)
    styled_pivot = pivot_table.style.background_gradient(cmap=cm).format("{:.2%}")

    return styled_pivot


# Main

In [11]:
table_results(urban=True)


Unnamed: 0_level_0,Unnamed: 1_level_0,Strategy,closest,random
Schedule Breaks,Prioritize Triage,Response Restricted,Unnamed: 3_level_1,Unnamed: 4_level_1
False,False,False,77.56%,21.15%
False,False,True,69.23%,23.72%
False,True,False,76.28%,19.23%
False,True,True,73.08%,22.44%
True,False,False,73.08%,22.44%
True,False,True,66.03%,18.59%
True,True,False,71.79%,14.10%
True,True,True,69.87%,17.31%


In [12]:
table_results(urban=False)


Unnamed: 0_level_0,Unnamed: 1_level_0,Strategy,closest,random
Schedule Breaks,Prioritize Triage,Response Restricted,Unnamed: 3_level_1,Unnamed: 4_level_1
False,False,False,90.00%,30.00%
False,False,True,80.00%,10.00%
False,True,False,90.00%,30.00%
False,True,True,90.00%,30.00%
True,False,False,80.00%,30.00%
True,False,True,80.00%,0.00%
True,True,False,80.00%,10.00%
True,True,True,90.00%,20.00%


In [13]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,64.18%,10.97,10.0
H,Yes,87.64%,21.62,20.28
V1,Yes,nan%,105.32,92.68
A,No,100.00%,15.92,14.05
H,No,50.00%,28.18,28.18
V1,No,nan%,63.53,63.53


In [14]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,62.69%,11.71,10.23
H,Yes,80.90%,23.65,21.38
V1,Yes,nan%,110.3,98.0
A,No,87.50%,17.57,15.38
H,No,50.00%,33.1,33.1
V1,No,nan%,64.08,64.08


In [15]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,49.25%,14.26,12.42
H,Yes,84.27%,21.65,19.18
V1,Yes,nan%,104.13,92.47
A,No,87.50%,17.39,15.32
H,No,50.00%,27.83,27.83
V1,No,nan%,63.37,63.37


In [16]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,49.25%,14.72,12.42
H,Yes,78.65%,22.86,20.65
V1,Yes,nan%,109.7,100.65
A,No,87.50%,17.35,15.41
H,No,50.00%,27.87,27.87
V1,No,nan%,63.82,63.82


In [17]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,67.16%,10.41,9.85
H,Yes,83.15%,22.45,19.83
V1,Yes,nan%,107.63,97.72
A,No,100.00%,14.15,13.13
H,No,50.00%,36.23,36.23
V1,No,nan%,63.83,63.83


In [18]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,62.69%,11.33,10.08
H,Yes,78.65%,23.37,20.9
V1,Yes,nan%,115.54,100.72
A,No,87.50%,16.41,14.02
H,No,50.00%,41.53,41.53
V1,No,nan%,64.15,64.15


In [19]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,62.69%,12.2,10.77
H,Yes,80.90%,22.19,18.43
V1,Yes,nan%,107.92,97.78
A,No,100.00%,15.55,14.15
H,No,50.00%,28.15,28.15
V1,No,nan%,63.63,63.63


In [20]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,55.22%,12.62,10.63
H,Yes,80.90%,23.16,20.4
V1,Yes,nan%,112.52,100.9
A,No,100.00%,14.8,14.15
H,No,50.00%,36.58,36.58
V1,No,nan%,63.95,63.95


In [21]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,13.43%,23.03,21.53
H,Yes,26.97%,42.45,38.07
V1,Yes,nan%,128.72,114.98
A,No,37.50%,39.01,29.77
H,No,0.00%,46.56,46.56
V1,No,nan%,93.22,93.22


In [22]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,19.40%,27.83,23.97
H,Yes,24.72%,44.91,39.87
V1,Yes,nan%,137.49,116.9
A,No,37.50%,36.34,26.85
H,No,0.00%,55.71,55.71
V1,No,nan%,114.33,114.33


In [23]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,13.43%,23.57,23.52
H,Yes,31.46%,42.49,41.4
V1,Yes,nan%,129.22,116.53
A,No,0.00%,37.52,35.6
H,No,50.00%,41.43,41.43
V1,No,nan%,115.18,115.18


In [24]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,10.45%,33.66,26.32
H,Yes,24.72%,45.73,42.15
V1,Yes,nan%,137.85,119.85
A,No,0.00%,37.57,36.37
H,No,0.00%,57.52,57.52
V1,No,nan%,126.43,126.43


In [25]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,11.94%,25.28,23.25
H,Yes,24.72%,44.63,38.78
V1,Yes,nan%,145.81,123.6
A,No,37.50%,31.93,33.0
H,No,0.00%,64.27,64.27
V1,No,nan%,95.53,95.53


In [26]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,5.97%,26.78,23.28
H,Yes,20.22%,46.15,38.13
V1,Yes,nan%,169.84,142.18
A,No,0.00%,44.93,42.08
H,No,50.00%,44.06,44.06
V1,No,nan%,88.63,88.63


In [27]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,7.46%,26.41,23.23
H,Yes,33.71%,42.23,36.07
V1,Yes,nan%,143.31,118.43
A,No,25.00%,35.29,38.53
H,No,50.00%,57.1,57.1
V1,No,nan%,98.05,98.05


In [28]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,5.97%,26.34,25.83
H,Yes,25.84%,45.15,41.07
V1,Yes,nan%,176.62,175.63
A,No,25.00%,36.33,38.93
H,No,0.00%,73.41,73.41
V1,No,nan%,98.47,98.47
