# Fix pathing

In [1]:
import sys


sys.path.append("../..")


In [2]:
import constants

import os


constants.PROJECT_DIRECTORY_PATH = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(constants.PROJECT_DIRECTORY_PATH))))


# Imports

In [3]:
import plotter
import datahandler

import matplotlib.pyplot
import numpy as np
import pandas as pd
import seaborn as sns
import IPython.display


# Constants

In [4]:
FOLDER_NAME = "2024_04_30_13_08_17_NONE"
DATE_BOUNDS = ("2017-08-10 07:00:00", "2017-08-10 18:59:59")

FOLDER_PATH = os.path.join(os.path.dirname(constants.PROJECT_DIRECTORY_PATH), "Simulator", "data", FOLDER_NAME)


In [5]:
data_preprocessor = datahandler.DataPreprocessorOUS_V2()
data_preprocessor.execute()

data_loader = datahandler.DataLoader(datahandler.DataPreprocessorOUS_V2)
data_loader.execute(False, False, True)


Cleaning dataset: 100%|██████████| 2/2 [00:00<00:00, 3990.77it/s]
Processing dataset: 100%|██████████| 2/2 [00:00<?, ?it/s]
Enhancing dataset: 100%|██████████| 2/2 [00:00<?, ?it/s]


Loading dataset: 100%|██████████| 2/2 [00:02<00:00,  1.47s/it]


# Methods

In [6]:
def load_csv(fileName):
    df = pd.read_csv(os.path.join(FOLDER_PATH, fileName + ".csv"))

    response_time_cols = [
        'duration_incident_creation',
        'duration_resource_appointment',
        'duration_resource_preparing_departure',
        'duration_dispatching_to_scene'
    ]
    df['total_response_time'] = df[response_time_cols].sum(axis=1)

    return df


In [7]:
def print_results(df: pd.DataFrame):
    # Define the criteria for response times
    criteria = {
        ('A', True): 12 * 60,  # 12 minutes converted to seconds
        ('A', False): 25 * 60,  # 25 minutes converted to seconds
        ('H', True): 30 * 60,  # 30 minutes converted to seconds
        ('H', False): 40 * 60  # 40 minutes converted to seconds
    }

    # Function to calculate compliance for each group
    def calculate_compliance(group, triage, urban):
        limit = criteria.get((triage, urban))
        if limit is not None:
            return (group['total_response_time'] < limit).mean()
        return None

    # Calculate statistics and compliance for each group
    results = []
    for (triage, urban), group in df.groupby(['triage_impression_during_call', 'urban']):
        mean = group['total_response_time'].mean()
        median = group['total_response_time'].median()
        compliance = calculate_compliance(group, triage, urban)
        results.append({
            'Triage': triage,
            'Urban': urban,
            'Mean (sec)': mean,
            'Median (sec)': median,
            'Compliance': compliance
        })

    stats = pd.DataFrame(results)

    # Convert mean and median to minutes
    stats['Mean (min)'] = (stats['Mean (sec)'] / 60)
    stats['Median (min)'] = (stats['Median (sec)'] / 60)
    stats.drop(columns=['Mean (sec)', 'Median (sec)'], inplace=True)

    # Map urban values to Yes/No
    stats['Urban'] = stats['Urban'].map({True: 'Yes', False: 'No'})
    
    # Sort values
    stats.sort_values(by=["Urban", "Triage"], ascending=[False, True], inplace=True)
    
    # Display the DataFrame
    formatted_stats = stats.style.format({
        'Mean (min)': "{:.2f}",
        'Median (min)': "{:.2f}",
        'Compliance': "{:.2%}"
    }).hide(axis='index')
    IPython.display.display(formatted_stats)


In [8]:
def boxplot_time_at_steps_modified(
    dataframe: pd.DataFrame,
    triage_impression: str = None
):
    title = "Time Taken At Each Step of the Incident"

    if triage_impression is not None:
        # Filter the dataframe without overwriting the original one
        temp_df = dataframe[dataframe["triage_impression_during_call"] == triage_impression].copy()
        title += f" ({triage_impression})"
    else:
        # Use the original dataframe if no triage_impression filter is applied
        temp_df = dataframe.copy()

    steps = {
        "Creating Incident": "duration_incident_creation",
        "Appointing Resource": "duration_resource_appointment",
        "Resource to Start Task": "duration_resource_preparing_departure",
        "Dispatching to Scene": "duration_dispatching_to_scene",
        "At Scene": "duration_at_scene",
        "Dispatching to Hospital": "duration_dispatching_to_hospital",
        "At Hospital": "duration_at_hospital"
    }

    # Calculating durations for each step
    plot_data = [(temp_df[duration_column][temp_df[duration_column] > 0] / 60) for step, duration_column in steps.items()]

    # Plotting
    matplotlib.pyplot.figure(figsize=(8, 4))
    matplotlib.pyplot.boxplot(plot_data[::-1], labels=list(steps.keys())[::-1], vert=False, patch_artist=True, showfliers=False)
    matplotlib.pyplot.title(title)
    matplotlib.pyplot.xlabel("Time in Minutes")
    matplotlib.pyplot.xticks()
    matplotlib.pyplot.show()


In [9]:
def boxplot_time_at_steps(
    historic_dataframe: pd.DataFrame,
    simulated_dataframe: pd.DataFrame,
    bounds: tuple[str, str] = None,
    triage_impressions: list[str] = ["A", "H", "V1"],
):
    historic_df = historic_dataframe.copy()
    simulated_df = simulated_dataframe.copy()
    
    # filter historic data by time frame
    if bounds is not None:
        start_bound, end_bound = pd.to_datetime(bounds[0]), pd.to_datetime(bounds[1])
        historic_df = historic_df[(historic_df['time_call_received'] >= start_bound) & (historic_df['time_call_received'] <= end_bound)]

    # calculate duration at each stage in minutes (simulated dataframe already has this calculated)
    historic_steps = {
        "duration_incident_creation": ("time_call_received", "time_incident_created"),
        "duration_resource_appointment": ("time_incident_created", "time_resource_appointed"),
        "duration_resource_preparing_departure": ("time_resource_appointed", "time_ambulance_dispatch_to_scene"),
        "duration_dispatching_to_scene": ("time_ambulance_dispatch_to_scene", "time_ambulance_arrived_at_scene"),
        "duration_at_scene": ("time_ambulance_arrived_at_scene", "time_ambulance_dispatch_to_hospital", "time_ambulance_available"),
        "duration_dispatching_to_hospital": ("time_ambulance_dispatch_to_hospital", "time_ambulance_arrived_at_hospital"),
        "duration_at_hospital": ("time_ambulance_arrived_at_hospital", "time_ambulance_available")
    }

    for step, times in historic_steps.items():
        if len(times) == 3:
            historic_df.loc[historic_df[times[1]].isna(), step] = (historic_df[times[2]] - historic_df[times[0]]).dt.total_seconds() / 60
            historic_df.loc[~historic_df[times[1]].isna(), step] = (historic_df[times[1]] - historic_df[times[0]]).dt.total_seconds() / 60
        else:
            historic_df[step] = (historic_df[times[1]] - historic_df[times[0]]).dt.total_seconds() / 60
    
    # convert seconds to minutes and replace zeros in simulated data to nan
    for step in historic_steps.keys():
        simulated_df[step] /= 60

    simulated_df[["duration_dispatching_to_hospital", "duration_at_hospital"]] = simulated_df[["duration_dispatching_to_hospital", "duration_at_hospital"]].replace(0, np.nan)

    # plot data
    matplotlib.pyplot.figure(figsize=(8, 8))

    data = []
    positions = []
    labels = []
    colors = []
    stripes = []

    i = 0
    for step in historic_steps.keys():
        labels.append(f"{step}")
        for triage in triage_impressions:
            for df, stripe, alpha in zip([historic_df, simulated_df], [False, True], [1.00, 0.65]):
                data.append(df[df['triage_impression_during_call'] == triage][step].dropna())

                labels.append("")

                if (triage == "A"):
                    colors.append([1.00, 0.37, 0.28, alpha])
                elif (triage == "H"):
                    colors.append([0.12, 0.56, 1.00, alpha])
                else:
                    colors.append([0.20, 0.80, 0.20, alpha])

                stripes.append(stripe)

                positions.append(i)
                i += 0.75
        labels.pop()
        i += 1

    bplot = matplotlib.pyplot.boxplot(
        data[::-1],
        labels=labels[::-1],
        positions=positions,
        vert=False,
        patch_artist=True,
        showfliers=True
    )

    for patch, color, stripe in zip(bplot["boxes"], colors[::-1], stripes[::-1]):
        patch.set_facecolor(color)

    matplotlib.pyplot.title("Time Taken At Each Step of the Incident")
    matplotlib.pyplot.xlabel("Time in Minutes")
    matplotlib.pyplot.xticks()
    matplotlib.pyplot.show()


In [48]:
def table_results(urban = True, strategies = ["closest", "random"]):
    def calculate_compliance(df):
        if df is None:
            return None

        # Criteria for both triage types in both urban and rural settings
        criteria = {
            ('A', True): 12 * 60,  # 12 minutes for triage 'A' in urban areas
            ('A', False): 25 * 60, # 25 minutes for triage 'A' in rural areas
            ('H', True): 30 * 60,  # 30 minutes for triage 'H' in urban areas
            ('H', False): 40 * 60  # 40 minutes for triage 'H' in rural areas
        }

        # Initialize counters
        total_compliant_cases = 0
        total_cases = 0
        
        # Check necessary columns exist
        if 'triage_impression_during_call' in df.columns and 'urban' in df.columns and 'total_response_time' in df.columns:
            for triage in ['A', 'H']:
                filtered_df = df[(df['triage_impression_during_call'] == triage) & (df['urban'] == urban)]
                limit = criteria.get((triage, urban))
                if not filtered_df.empty:
                    # Count compliant cases for this triage type
                    compliant_cases = filtered_df['total_response_time'] < limit
                    total_compliant_cases += compliant_cases.sum()
                    total_cases += len(filtered_df)
            if total_cases > 0:
                # Calculate overall compliance rate across both triage types
                overall_compliance = total_compliant_cases / total_cases
                return overall_compliance
            else:
                return None  # Return None if there are no cases to evaluate
        else:
            # If required columns are missing, return None to indicate an issue with the data format.
            return None

    # Settings combinations
    prioritize_triages = [False, True]
    response_restricteds = [False, True]
    schedule_breaks = [False, True]

    results = []

    for strategy in strategies:
        for prioritize_triage in prioritize_triages:
            for response_restricted in response_restricteds:
                for schedule_break in schedule_breaks:
                    filename = f"events_strategy={strategy}_prioritizeTriage={'true' if prioritize_triage else 'false'}_responseRestricted={'true' if response_restricted else 'false'}_scheduleBreaks={'true' if schedule_break else 'false'}"
                    df = load_csv(filename)
                    compliance = calculate_compliance(df)
                    if compliance is not None:
                        results.append({
                            "Strategy": strategy,
                            "Prioritize Triage": prioritize_triage,
                            "Response Restricted": response_restricted,
                            "Schedule Breaks": schedule_break,
                            "Compliance": compliance
                        })

    # Create DataFrame
    results_df = pd.DataFrame(results)
    # Pivot Table for visualization
    pivot_table = results_df.pivot_table(index=["Schedule Breaks", "Prioritize Triage", "Response Restricted"],
                                        columns=["Strategy"], 
                                        values="Compliance")
    # Color coding from green to red
    cm = sns.light_palette("green", as_cmap=True, n_colors=8)
    styled_pivot = pivot_table.style.background_gradient(cmap=cm).format("{:.2%}")

    return styled_pivot


# Main

In [49]:
table_results(urban=True)


Unnamed: 0_level_0,Unnamed: 1_level_0,Strategy,closest,random
Schedule Breaks,Prioritize Triage,Response Restricted,Unnamed: 3_level_1,Unnamed: 4_level_1
False,False,False,79.04%,23.95%
False,False,True,74.85%,16.77%
False,True,False,77.84%,25.75%
False,True,True,74.85%,20.96%
True,False,False,78.44%,22.16%
True,False,True,71.26%,23.95%
True,True,False,76.05%,19.76%
True,True,True,71.86%,19.76%


In [50]:
table_results(urban=False)


Unnamed: 0_level_0,Unnamed: 1_level_0,Strategy,closest,random
Schedule Breaks,Prioritize Triage,Response Restricted,Unnamed: 3_level_1,Unnamed: 4_level_1
False,False,False,92.86%,42.86%
False,False,True,78.57%,35.71%
False,True,False,92.86%,21.43%
False,True,True,92.86%,28.57%
True,False,False,92.86%,14.29%
True,False,True,78.57%,28.57%
True,True,False,92.86%,7.14%
True,True,True,92.86%,14.29%


In [11]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,75.32%,10.29,9.88
H,Yes,82.22%,22.86,20.95
V1,Yes,nan%,71.86,62.63
A,No,100.00%,14.16,13.9
H,No,80.00%,25.61,28.37
V1,No,nan%,54.28,56.4


In [12]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,74.03%,10.82,10.13
H,Yes,82.22%,23.35,21.12
V1,Yes,nan%,73.22,67.15
A,No,100.00%,15.37,14.08
H,No,80.00%,26.96,28.3
V1,No,nan%,54.01,55.77


In [13]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,61.04%,12.31,10.68
H,Yes,86.67%,22.35,20.6
V1,Yes,nan%,71.89,62.55
A,No,77.78%,17.61,14.08
H,No,80.00%,25.75,28.47
V1,No,nan%,54.09,55.98


In [14]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,54.55%,13.2,11.08
H,Yes,85.56%,23.06,20.84
V1,Yes,nan%,72.5,62.57
A,No,77.78%,17.91,14.08
H,No,80.00%,27.18,29.9
V1,No,nan%,54.2,56.17


In [15]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,74.03%,10.19,9.42
H,Yes,81.11%,23.34,21.16
V1,Yes,nan%,73.55,63.42
A,No,100.00%,13.97,13.85
H,No,80.00%,25.65,28.42
V1,No,nan%,54.16,56.06


In [16]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,71.43%,10.42,9.92
H,Yes,80.00%,24.18,21.63
V1,Yes,nan%,73.79,67.2
A,No,100.00%,14.45,13.83
H,No,80.00%,27.21,28.55
V1,No,nan%,54.01,55.82


In [17]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,67.53%,11.19,10.3
H,Yes,81.11%,23.08,20.73
V1,Yes,nan%,73.5,63.45
A,No,100.00%,15.62,14.4
H,No,80.00%,25.53,28.45
V1,No,nan%,54.06,55.93


In [18]:
print_results(load_csv("events_strategy=closest_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,59.74%,11.82,10.77
H,Yes,82.22%,23.74,21.74
V1,Yes,nan%,78.79,69.3
A,No,100.00%,14.85,13.95
H,No,80.00%,26.85,28.28
V1,No,nan%,54.14,56.09


In [19]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,11.69%,27.27,26.32
H,Yes,34.44%,38.9,36.24
V1,Yes,nan%,95.17,87.38
A,No,55.56%,29.37,23.08
H,No,20.00%,57.57,53.18
V1,No,nan%,98.95,104.12


In [20]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,10.39%,27.47,25.7
H,Yes,32.22%,39.92,35.87
V1,Yes,nan%,98.62,88.83
A,No,11.11%,40.65,38.78
H,No,20.00%,48.95,45.37
V1,No,nan%,99.11,95.46


In [21]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,9.09%,25.72,24.55
H,Yes,23.33%,41.52,40.21
V1,Yes,nan%,99.18,86.0
A,No,44.44%,33.17,26.58
H,No,20.00%,52.76,54.77
V1,No,nan%,91.3,87.97


In [22]:
print_results(load_csv("events_strategy=random_prioritizeTriage=false_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,12.99%,25.1,22.32
H,Yes,33.33%,42.22,38.3
V1,Yes,nan%,97.17,87.42
A,No,22.22%,36.91,32.1
H,No,40.00%,50.28,49.75
V1,No,nan%,102.38,101.56


In [23]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,16.88%,25.73,24.03
H,Yes,33.33%,39.49,35.52
V1,Yes,nan%,101.46,95.23
A,No,33.33%,35.19,33.42
H,No,0.00%,60.59,60.55
V1,No,nan%,152.95,134.27


In [24]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=false_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,11.69%,26.92,26.88
H,Yes,26.67%,46.72,40.33
V1,Yes,nan%,110.67,122.07
A,No,0.00%,40.2,39.15
H,No,20.00%,62.16,66.12
V1,No,nan%,142.95,143.36


In [25]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=false"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,14.29%,23.52,21.3
H,Yes,26.67%,43.67,41.73
V1,Yes,nan%,109.85,86.55
A,No,22.22%,39.88,40.73
H,No,40.00%,56.12,70.37
V1,No,nan%,112.72,108.83


In [26]:
print_results(load_csv("events_strategy=random_prioritizeTriage=true_responseRestricted=true_scheduleBreaks=true"))


Triage,Urban,Compliance,Mean (min),Median (min)
A,Yes,6.49%,25.23,23.32
H,Yes,31.11%,45.37,36.86
V1,Yes,nan%,116.89,125.82
A,No,11.11%,40.08,40.78
H,No,20.00%,56.1,46.27
V1,No,nan%,138.53,121.81
