# 2: Interacting with a CSV Data

## Setup and connect to the Azure OpenAI endpoint

In [28]:
import os 
import pandas as pd

from IPython.display import Markdown, HTML, display
from langchain.schema import HumanMessage
from langchain_openai import AzureChatOpenAI
from langchain_experimental.tools import PythonAstREPLTool

# Ensure environment variables are set
AZURE_OPENAI_API_VERSION = os.environ["AZURE_OPENAI_API_VERSION"]
AZURE_OPENAI_MODEL = os.environ["AZURE_OPENAI_MODEL"]
AZURE_OPENAI_ENDPOINT = os.environ["AZURE_OPENAI_ENDPOINT"]
AZURE_OPENAI_API_KEY = os.environ["AZURE_OPENAI_API_KEY"]

# Connect to the Azure OpenAI endpoint
model = AzureChatOpenAI(
    openai_api_version=AZURE_OPENAI_API_VERSION,
    azure_deployment=AZURE_OPENAI_MODEL,
    azure_endpoint=AZURE_OPENAI_ENDPOINT,
    openai_api_key=AZURE_OPENAI_API_KEY
)

## Load the dataset

In [29]:
df = pd.read_csv("./data/all-states-history.csv").fillna(value = 0)

## Prepare the Langchain dataframe agent

In [30]:
from langchain.agents.agent_types import AgentType
from langchain_experimental.agents.agent_toolkits import create_pandas_dataframe_agent

agent = create_pandas_dataframe_agent(
    llm=model,
    df=df,
    verbose=True,
    agent_type=AgentType.OPENAI_FUNCTIONS
)


agent.invoke("how many rows are there? ")



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `python_repl_ast` with `{'query': 'len(df)'}`


[0m[36;1m[1;3m20780[0m[32;1m[1;3mThe dataframe `df` contains 20,780 rows.[0m

[1m> Finished chain.[0m


{'input': 'how many rows are there? ',
 'output': 'The dataframe `df` contains 20,780 rows.'}

## Design your prompt and ask your question

In [31]:
CSV_PROMPT_PREFIX = """
First set the pandas display options to show all the columns,
get the column names, then answer the question.
"""

CSV_PROMPT_SUFFIX = """
- **ALWAYS** before giving the Final Answer, try another method.
Then reflect on the answers of the two methods you did and ask yourself
if it answers correctly the original question.
If you are not sure, try another method.
- If the methods tried do not give the same result,reflect and
try again until you have two methods that have the same result.
- If you still cannot arrive to a consistent result, say that
you are not sure of the answer.
- If you are sure of the correct answer, create a beautiful
and thorough response using Markdown.
- **DO NOT MAKE UP AN ANSWER OR USE PRIOR KNOWLEDGE,
ONLY USE THE RESULTS OF THE CALCULATIONS YOU HAVE DONE**.
- **ALWAYS**, as part of your "Final Answer", explain how you got
to the answer on a section that starts with: "\n\nExplanation:\n".
In the explanation, mention the column names that you used to get
to the final answer.
"""

QUESTION = "How may patients were hospitalized during July 2020" 
"in Texas, and nationwide as the total of all states?"
"Use the hospitalizedIncrease column" 


agent.invoke(CSV_PROMPT_PREFIX + QUESTION + CSV_PROMPT_SUFFIX)



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `functions` with `{'query': "import pandas as pd\npd.set_option('display.max_columns', None)\ndf.columns.tolist()"}`


[0mfunctions is not a valid tool, try one of [python_repl_ast].[32;1m[1;3m
Invoking: `python_repl_ast` with `{'query': "import pandas as pd\npd.set_option('display.max_columns', None)\ndf.columns.tolist()"}`


[0m[36;1m[1;3m['date', 'state', 'death', 'deathConfirmed', 'deathIncrease', 'deathProbable', 'hospitalized', 'hospitalizedCumulative', 'hospitalizedCurrently', 'hospitalizedIncrease', 'inIcuCumulative', 'inIcuCurrently', 'negative', 'negativeIncrease', 'negativeTestsAntibody', 'negativeTestsPeopleAntibody', 'negativeTestsViral', 'onVentilatorCumulative', 'onVentilatorCurrently', 'positive', 'positiveCasesViral', 'positiveIncrease', 'positiveScore', 'positiveTestsAntibody', 'positiveTestsAntigen', 'positiveTestsPeopleAntibody', 'positiveTestsPeopleAntigen', 'positiveTestsViral', 'recover

A value is trying to be set on a copy of a slice from a DataFrame.
Try using .loc[row_indexer,col_indexer] = value instead

See the caveats in the documentation: https://pandas.pydata.org/pandas-docs/stable/user_guide/indexing.html#returning-a-view-versus-a-copy


[32;1m[1;3m
Invoking: `python_repl_ast` with `{'query': "al_july_2020_totals = al_data[(al_data['date'] >= '2020-07-01') & (al_data['date'] <= '2020-07-31')]\nal_july_2020_totals"}`


[0m[36;1m[1;3m            date state   death  deathConfirmed  deathIncrease  deathProbable  \
12265 2020-07-31    AL  1580.0          1531.0             15           49.0   
12321 2020-07-30    AL  1565.0          1516.0             27           49.0   
12377 2020-07-29    AL  1538.0          1489.0             47           49.0   
12433 2020-07-28    AL  1491.0          1446.0              0           45.0   
12489 2020-07-27    AL  1491.0          1446.0             18           45.0   
12545 2020-07-26    AL  1473.0          1428.0             17           45.0   
12601 2020-07-25    AL  1456.0          1413.0             18           43.0   
12657 2020-07-24    AL  1438.0          1395.0             41           43.0   
12713 2020-07-23    AL  1397.0          1357.0             33           40.0 

{'input': '\nFirst set the pandas display options to show all the columns,\nget the column names, then answer the question.\nHow may patients were hospitalized during July 2020\n- **ALWAYS** before giving the Final Answer, try another method.\nThen reflect on the answers of the two methods you did and ask yourself\nif it answers correctly the original question.\nIf you are not sure, try another method.\n- If the methods tried do not give the same result,reflect and\ntry again until you have two methods that have the same result.\n- If you still cannot arrive to a consistent result, say that\nyou are not sure of the answer.\n- If you are sure of the correct answer, create a beautiful\nand thorough response using Markdown.\n- **DO NOT MAKE UP AN ANSWER OR USE PRIOR KNOWLEDGE,\nONLY USE THE RESULTS OF THE CALCULATIONS YOU HAVE DONE**.\n- **ALWAYS**, as part of your "Final Answer", explain how you got\nto the answer on a section that starts with: "\n\nExplanation:\n".\nIn the explanation, 