In [1]:
%%time
import pal
import json
import random
import io
import pandas as pd
from datetime import datetime
from dateutil.relativedelta import relativedelta, MO
from pal.prompt import math_prompts, colored_object_prompt, penguin_prompt, date_understanding_prompt, algorithmic_prompt
from tqdm.notebook import tqdm

CPU times: user 574 ms, sys: 40.1 ms, total: 614 ms
Wall time: 903 ms


In [2]:
mixed_questions = [
    {"input": """Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Freeman, 43, 25, 69
Owen, 28, 61, 42
Jack, 11, 23, 83
For example: The height of Owen is 61 cm. What is the product of the heights of the two youngest penguins?
""".strip(), "target": 1403},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, birthday, height (cm), weight (kg)
Pops, 5/10/2006, 18, 6
Icle, 6/12/2003, 21, 8
Stik, 9/23/2015, 41, 12
For example: The weight of Stik is 12 kg. Today is 8/20/2020. What is the product of the heights of all penguins born over 8 years ago?
""".strip(), "target": 378},
    {"input": """Here is a table where the first line is a header and each subsequent line is a bug:
name, height (cm), weight (kg)
Bobby, 18, 13
Jacky, 17, 17
Smith, 20, 4
For example: The height of Bobby is 18 cm. What is the sum of the weights of the two tallest bugs?
""".strip(), "target": 17},
    {"input": """Today is May 3rd, 1972. I was born exactly 7 years ago. What is the sum of the year I was born and the day I was born?
    """.strip(), "target": 1968},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, birthday, height (cm), weight (kg)
Pops, 5/10/2006, 18, 6
Icle, 6/12/2003, 21, 8
Stik, 9/23/2015, 41, 12
For example: The weight of Stik is 12 kg. Today is 6/12/2016. What is the sum of the weight and height of the octopus who was born exactly 13 years ago?
    """.strip(), "target": 29},
    {"input": """Here is a table where the first line is a header and each subsequent line is a bug:
name, height (cm), weight (kg)
Bobby, 18, 13
Jacky, 17, 17
Smith, 20, 4
For example: The height of Bobby is 18 cm. What is the product of the weights of all three bugs?
    """.strip(), "target": 884},
    {"input": """Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Freeman, 43, 25, 69
Owen, 28, 61, 42
Jack, 11, 23, 83
For example: The height of Owen is 61 cm. Find the (positive) difference between the age of the tallest penguin and the age of the shortest penguin.
    """.strip(), "target": 17},
    {"input": """Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Fren, 43, 25, 69
Owen, 28, 61, 42
Jack, 11, 23, 83
Cobb, 16, 12, 53
Lobe, 93, 10, 99
For example: The height of Owen is 61 cm. What is the sum of the heights of all penguins who are older than 20?
    """.strip(), "target": 96},
    {"input": """Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Fren, 43, 25, 69
Owen, 28, 61, 42
Jack, 11, 23, 83
Cobb, 16, 12, 53
Lobe, 93, 10, 99
For example: The height of Owen is 61 cm. What is the product of the weights of all penguins who are younger than 20?
    """.strip(), "target": 4399},
    {"input": """5 months ago, the date was 9/23/2008. What is the sum of the year and the day today?
    """.strip(), "target": 2032},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, height (cm), weight (kg)
Pops, 18, 6
Icle, 21, 8
Stik, 41, 12
Blob, 23, 9
Kobe, 99, 18
For example: The weight of Stik is 12 kg. What is the result when the weight of the tallest octopus is divided by the weight of the shortest octopus?
    """.strip(), "target": 3},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, height (cm), weight (kg)
Pops, 18, 18
Icle, 21, 8
Stik, 41, 18
Blob, 23, 9
Kobe, 99, 18
For example: The weight of Stik is 18 kg. What is the sum of the heights of all octopus that weigh 18 kg?
    """.strip(), "target": 158},
    {"input": """Here is a table where the first line is a header and each subsequent line is a bug:
name, height (cm), weight (kg), x-coordinate
Bobby, 18, 13, 13
Jacky, 17, 17, 82
Smith, 20, 42, 18
Jonny, 33, 23, 35
For example: The weight of Bobby is 13 kg. What is the sum of the heights of the leftmost and rightmost bugs based on x-coordinate?
    """.strip(), "target": 35},
    {"input": """Here is a table where the first line is a header and each subsequent line is a bug:
name, height (cm), weight (kg), x-coordinate
Bobby, 18, 13, 13
Jacky, 17, 17, 82
Smith, 20, 42, 18
Jonny, 33, 23, 35
For example: The weight of Bobby is 13 kg. What is the sum of the height and weight of the bug with x-coordinate equal to 18?
    """.strip(), "target": 62},
    {"input": """Here is a table where the first line is a header and each subsequent line is a person:
name, height (cm), weight (kg), job
Abe, 193, 58, Doctor
Row, 116, 16, Lawyer
Bay, 152, 46, Doctor
Cho, 123, 25, Gardener
For example: The height of Row is 116 cm. What is the average weight of the doctors?
    """.strip(), "target": 52},
    {"input": """Here is a table where the first line is a header and each subsequent line is a person:
name, height (cm), weight (kg), job
Abe, 193, 58, Doctor
Row, 116, 16, Lawyer
Bay, 152, 46, Doctor
Cho, 123, 25, Gardener
For example: The height of Row is 116 cm. How many kilograms is Cho heavier than Row?
    """.strip(), "target": 9},
    {"input": """Here is a table where the first line is a header and each subsequent line is a person:
name, height (cm), weight (kg), job
Abe, 193, 58, Doctor
Row, 116, 16, Lawyer
Bay, 152, 46, Doctor
Cho, 123, 25, Gardener
For example: The height of Row is 116 cm. What is the sum of the height of the lawyer and the weight of the gardener?
    """.strip(), "target": 141},
    {"input": """Here is a table where the first line is a header and each subsequent line is a person:
name, height (cm), weight (kg), job
Abe, 193, 58, Doctor
Row, 116, 16, Lawyer
Bay, 152, 46, Doctor
Cho, 123, 25, Gardener
For example: The height of Row is 116 cm. What is the product of the weights of the doctors?
    """.strip(), "target": 2668},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, height (cm), weight (kg)
Pops, 18, 18
Icle, 21, 8
Stik, 41, 18
Blob, 23, 9
Kobe, 99, 9
For example: The weight of Stik is 18 kg. What is the product of the heights of all penguins whose weight is larger than 9 kg?
    """.strip(), "target": 738},
    {"input": """Here is a table where the first line is a header and each subsequent line is an octopus:
name, height (cm), weight (kg)
Pops, 18, 16
Icle, 21, 8
Stik, 41, 12
Blob, 23, 9
Kobe, 99, 20
For example: The weight of Stik is 12 kg. Find the sum of the heights of the lightest and heaviest penguins.
    """.strip(), "target": 120}
                  ]

In [3]:
for i in range(len(mixed_questions)):
    print("Problem: " + str(i + 1))
    print("Question: " + mixed_questions[i]["input"])
    print("Answer: " + str(mixed_questions[i]["target"]))
    print("")

Problem: 1
Question: Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Freeman, 43, 25, 69
Owen, 28, 61, 42
Jack, 11, 23, 83
For example: The height of Owen is 61 cm. What is the product of the heights of the two youngest penguins?
Answer: 1403

Problem: 2
Question: Here is a table where the first line is a header and each subsequent line is an octopus:
name, birthday, height (cm), weight (kg)
Pops, 5/10/2006, 18, 6
Icle, 6/12/2003, 21, 8
Stik, 9/23/2015, 41, 12
For example: The weight of Stik is 12 kg. Today is 8/20/2020. What is the product of the heights of all penguins born over 8 years ago?
Answer: 378

Problem: 3
Question: Here is a table where the first line is a header and each subsequent line is a bug:
name, height (cm), weight (kg)
Bobby, 18, 13
Jacky, 17, 17
Smith, 20, 4
For example: The height of Bobby is 18 cm. What is the sum of the weights of the two tallest bugs?
Answer: 17

Problem: 4
Question: 

In [26]:
interface = pal.interface.ProgramInterface(
  model='gpt-4o-mini',
  stop='\n\n\n', # stop generation str for Codex API
  get_answer_expr='solution()' # python expression evaluated after generated code to obtain answer 
)
libraryPrompt = """There are 3 kinds of libraries you can use:
(1) Arithmetic - This contains functions that allow you to perform arithmetic operations
(2) Dates - This contains functions that allow you to perform operations on calendar dates
(3) Tables - This contains functions that allow you to perform operations on tables.
To choose a library to use, write CHOOSE_LIBRARY[library_name]. The list of functions in the library will be provided to you. For example, if you want to use Dates, write CHOOSE_LIBRARY[Dates].
"""
reminderPrompt = """Reminder that you can choose another library to use, if needed, by writing CHOOSE_LIBRARY[library_name]. IMPORTANT NOTE: Once you switch to another library, you cannot use the functions of the previous libraries."""
allPrompts = ["""Whenever you synthesize information and perform reasoning, it is a Thought. When you need to calculate a value, e.g., 5+3, it is an Action. There are 5 types of actions: 
(1)ADD[a;b]: ADD takes in two numbers a and b, and returns their sum. For example, ADD[7;3] will return the value of 7+3.
(2)SUB[a;b]: SUB takes in two numbers a and b, and returns their difference. For example, SUB[6;2] will return the value of 6-2.
(3)MULT[a;b]: MULT takes in two numbers a and b, and returns their product. For example, MULT[8;5] will return the value of 8*5.
(4)DIV[a;b]: DIV takes in two numbers a and b, and returns their quotient. For example, DIV[9;3] will return 9/3.
(5)FINISH_START[answer]FINISH_END: When you finish a question, write your answer as FINISH_START[answer]FINISH_END and terminate your response.
Terminate your response after an action. Then, an Observation is returned by the user. IMPORTANT: YOU DO NOT WRITE OBSERVATIONS. THEY WILL BE PROVIDED TO YOU. You will be given a question and you will perform Thoughts and Actions until you get the answer. When you get the answer, you will perform the Action called Finish and write Finish[answer]. While you are reasoning out solutions, you will continually refer back to relevant portions of the question and employ logical reasoning to determine your next action.""",  
              """Whenever you synthesize information and perform reasoning, it is a Thought. When you need to find a date, it is an Action. There are 6 types of actions: incYear[M;D;Y;n], incMonth[M;D;Y;n], incDay[M;D;Y;n], decYear[M;D;Y;n], decMonth[M;D;Y;n], decDay[M;D;Y;n]. M, D, and Y are the Month, Day, and Year as an integer, respectively. n is the number of Years/Months/Days that you want to increment/decrement by. For example, if you wish to find the date 2 months after May 13th, 1973, you would perform the action incMonth[5, 13, 1973, 2]. Terminate your response after an action. Then, an Observation is returned by the user. IMPORTANT: YOU DO NOT WRITE OBSERVATIONS. THEY WILL BE PROVIDED TO YOU. You will be given a question and you will perform Thoughts and Actions until you get the answer. When you get the answer, you will perform the Action called Finish and write Finish[answer]. While you are reasoning out solutions, you will continually refer back to relevant portions of the question and employ logical reasoning to determine your next action.""", 
              """Whenever you synthesize information and perform reasoning, it is a Thought. When you need to perform an operation on a table, it is an Action. Here is an example table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Feynman, 43, 25, 69
Oppie, 28, 73, 42
Bob, 11, 23, 83
There are 7 types of actions: 
(1) CREATE_START[data]CREATE_END: CREATE takes a data definition of a table which will be used to generate a pandas dataframe, so make sure that the input does not contain anything unnecessary for the pandas dataframe input generation. Write the data in json format. For example, to create a table for the above example, you would write: CREATE_START[{"name": ["Louis", "Bernard", "Vincent", "Gwen"], "age": [7, 5, 9, 8], "height": [50, 80, 60, 70], "weight": [11, 13, 11, 15]}]CREATE_END
(2) GET_VALUE_START[row_number;column]GET_VALUE_END: GET_VALUE takes the row_number, e.g., 0 is the 1st row of the table, and the name of a column, e.g., age, and returns the value of the cell, e.g., the age in the first row of the table. For example, to get the age in the first row of the table, you would rate GET_VALUE_START[0;age]
(3) COUNT_START[query]COUNT_END: COUNT takes a query and counts the number of entries in the table that satisfies the query. For example, COUNT_START[age > 5]COUNT_END would count the number of entries whose value in the column "age" is larger than 5.
(4) FILTER_START[query]FILTER_END: FILTER takes a query, similarly to COUNT, and filters the entries in the table that satisfies the query. For example, FILTER_START[height < 13]FILTER_END would return the entries whose value in the column "height" is less than 13.
(5) SORT_START[column]SORT_END: SORT takes the name of a column in the table and sorts the table based on the column in lexicographical/numerical order. For example, in the above table, SORT_START[age]SORT_START would sort the table from the youngest penguin to the oldest penguin.
(6) ADD_START[row]ADD_END: ADD takes a row and adds it to the table. Write the data in json format. NOTE: This does not necessarily preserve any previous sorted order. For example, to add a penguin named Jack, age 8, height 52 cm, and weight 18 kg, you would write ADD_START[{"name": "Jack", "age": 8, "height": 52, "weight": 18}]ADD_END
(7) FINISH_START[answer]FINISH_END: When you finish the question, write your answer as FINISH_START[answer]FINISH_END and terminate your response.
Terminate your response after an action. Then, an Observation is returned by the user. IMPORTANT: YOU DO NOT WRITE OBSERVATIONS. THEY WILL BE PROVIDED TO YOU. You will be given a question and you will perform Thoughts and Actions until you get the answer. When you get the answer, you will perform the Action called Finish and write Finish[answer]. While you are reasoning out solutions, you will continually refer back to relevant portions of the question and employ logical reasoning to determine your next action.
"""]

allMessages = [
    [
        {"role": "system", "content": "You are a helpful assistant that accurately solves mathematical reasoning questions by acting as an agent who will perform Thoughts and Actions."},
        {"role": "user", "content": allPrompts[0] + '\n' + "Lisa is looking to attempt a World Record. She has decided to try and match Joey Chestnut's record of eating 75 full hotdogs, buns included, in 10 minutes. Halfway through the time Lisa has eaten 20 hotdogs. How many hotdogs will she have to eat per minute to at least tie Joey Chestnut's record?"},
        {"role": "assistant", "content": """Thought: Joey Chestnut ate 75 hotdogs to claim the record and Lisa has eaten 20 hot dogs so far, so she still needs to eat 75 - 20 hotdogs to tie Joey Chestnut.
Action: SUB[75;20]"""},
        {"role": "user", "content": "Observation: 55"},
        {"role": "assistant", "content": """Thought: Lisa still has to eat 55 hotdogs. Lisa has a 10 minute time period to eat the hotdogs and half the time has already passed, which means Lisa has 10/2 minutes left.
Action: DIV[10;2]"""},
        {"role": "user", "content": "Observation: 5"},
        {"role": "assistant", "content": """Thought: So Lisa has 5 minutes left to eat 55 hotdogs. Therefore, she needs to eat 55/5 hotdogs per minute to tie.
Action: DIV[55;5]"""},
        {"role": "user", "content": "Observation: 11"},
        {"role": "assistant", "content": """Thought: Therefore, Lisa has to eat 11 hotdogs per minute to tie Joey Chestnut's record.
Action: FINISH[11]"""},
#        {"role": "user", "content": allPrompts[0] + '\n' + "Joseph and his friends watched two movies in his house. The first movie is 1 hour and 30 minutes long while the second movie is 30 minutes longer than the first. Before the movies, they spent 10 minutes making popcorn and twice as long making fries. How long, in hours, did it take Joseph and his friends to cook and watch the movies?"},
#        {"role": "assistant", "content": """Thought: An hour is 60 minutes, so the first movie was 60 + 30 minutes long.
#Action: ADD[60;30]"""},
#        {"role": "user", "content": "Observation: 90"},
#        {"role": "assistant", "content": """Thought: So the first movie was 90 minutes long. Therefore, the second movie is 90 + 30 minutes long.
#Action: ADD[90 + 30]"""},
#        {"role": "user", "content": "Observation: 120"},
#        {"role": "assistant", "content": """Thought: So the second movie is 120 minutes long. It took them a total of 90 + 120 minutes to watch the two movies.
#Action: ADD[90;120]"""},
#        {"role": "user", "content": "Observation: 210"},
#        {"role": "assistant", "content": """Thought: Thus, it took them 210 minutes to watch the two movies. It also took twice as long making fries as making popcorn, and since it took 10 minutes to make popcorn, it took 10 * 2 minutes to make fries.
#Action: MULT[10;2]"""},
#        {"role": "user", "content": "Observation: 20"},
#        {"role": "assistant", "content": """Thought: Therefore, it took 20 minutes to make the fries. Since it took 10 minutes to make the popcorn, in total, it took 210 + 20 + 10 minutes to cook and watch the movies. We first find 210 + 20.
#Action: ADD[210;20]"""},
#        {"role": "user", "content": "Observation: 230"},
#        {"role": "assistant", "content": """Thought: Therefore, 210 + 20 + 10 equals 230 + 10. Now we find 230 + 10.
#Action: ADD[230;10]"""},
#        {"role": "user", "content": "Observation: 240"},
#        {"role": "assistant", "content": """Thought: Thus, it took 240 minutes in total to cook and watch the movies. Finally, the problem asks for how many hours it took. Each hour is 60 minutes, so we divide 240 by 60 to get the number of hours.
#Action: DIV[240;60]"""},
#        {"role": "user", "content": "Observation: 4"},
#        {"role": "assistant", "content": """Thought: So it took 4 hours in total to cook and watch the movies.
#Action: FINISH[4]"""}
    ],
    [
        {"role": "system", "content": "You are a helpful assistant that accurately solves reasoning questions about dates by acting as an agent who will perform Thoughts and Actions."},
        {"role": "user", "content": allPrompts[1] + '\n' + "Question: In 5 months, the date will be September 18th, 2016. What will be the date in 53 days?"},
        {"role": "assistant", "content": """Thought: If the date in 5 months will be September 18th, 2016, then today is the date 5 months before 9/18/2016. Thus, I need to find the date 5 months before 9/18/2016. 
Action: decMonth[9;18;2016;5]"""},
        {"role": "user", "content": "Observation: 4/18/2016"},
        {"role": "assistant", "content": """Thought: So today is 4/18/2016. We want to find the date in 53 days.
Action: incDay[4;18;2016;53]"""},
        {"role": "user", "content": "Observation: 6/10/2016"},
        {"role": "assistant", "content": """Thought: So in 53 days, the date will be 6/10/2016.
Action: FINISH[06/10/2016]"""}
    ],
    [
        {"role": "system", "content": "You are a helpful assistant that accurately solves reasoning questions about tables by acting as an agent who will perform Thoughts, where you will perform logical reasoning, and Actions, which act on the tables."},
        {"role": "user", "content": allPrompts[2] + '\n' + """Question: Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Louis, 7, 50, 11
Bernard, 5, 80, 13
Vincent, 9, 60, 11
Gwen, 8, 70, 15
For example: the age of Louis is 7, the weight of Gwen is 15 kg, the height of Bernard is 80 cm. 
We now add a penguin to the table:
James, 12, 90, 12
How many penguins are less than 8 years old?"""},
        {"role": "assistant", "content": """Thought: I first need to create the table.
Action: CREATE_START[{"name": ["Louis", "Bernard", "Vincent", "Gwen"], "age": [7, 5, 9, 8], "height": [50, 80, 60, 70], "weight": [11, 13, 11, 15]}]CREATE_END
"""},
        {"role": "user", "content": "Observation: Created table"},
        {"role": "assistant", "content": """Thought: Now I need to add a new row for the new penguin
Action: ADD_START[{"name": "James", "age": 12, "height": 90, "weight": 12}]ADD_END"""},
        {"role": "user", "content": "Observation: Added row"},
        {"role": "assistant", "content": """Thought: Finally, we want to find the number of penguins whose age is less than 8.
Action: COUNT_START[age < 8]COUNT_END"""},
        {"role": "user", "content": "Observation: 2"},
        {"role": "assistant", "content": """Thought: So there are 2 penguins whose age is less than 8.
#Action: FINISH_START[2]FINISH_END"""},
#        {"role": "user", "content": allPrompts[2] + '\n' + """Question: Here is a table where the first line is a header and each subsequent line is a penguin:
#name, age, height (cm), weight (kg) 
#Feynman, 43, 25, 69
#Oppie, 28, 73, 42
#Bob, 11, 23, 83
#For example: The height of Oppie is 73 cm. What is the name of the oldest penguin?"""},
#        {"role": "assistant", "content": """Thought: I first need to create the table
#Action: CREATE_START[{"name": ["Feynman", "Oppie", "Bob"], "age": [43, 28, 11], "height": [25, 73, 23], "weight": [69, 42, 83]}]CREATE_END"""},
#        {"role": "user", "content": "Observation: Created table"},
#        {"role": "assistant", "content": """Thought: I want the oldest penguin, so first I need to sort the table by age.
#Action: SORT_START[age]SORT_END"""},
#        {"role": "user", "content": "Observation: Sorted table by age"},
#        {"role": "assistant", "content": """Thought: Since I sorted the table, the oldest penguin must be the third row. Furthermore, I need to find their name.
#Action: GET_VALUE_START[2;name]"""},
#        {"role": "user", "content": "Observation: Feynman"},
#        {"role": "assistant", "content": """Thought: So the oldest penguin is named Feynman.
#Action: FINISH_START[Feynman]FINISH_END"""}
    ]
           ]

json_list = []
max_count = 5
counter = 0
responses = []
#for entry in mixed_questions:
#    json_list.append(entry)
with open('datasets/svamp.jsonl', 'r') as json_file:
    allEntries = list(json_file)
    for entry in allEntries:
        result = json.loads(entry)
        json_list.append(result)
#with open('datasets/date_understanding.json', 'r') as json_file:
#    allEntries = json.load(json_file)
#    allEntries = allEntries["examples"]
#    for entry in allEntries:
#        json_list.append(entry)
#with open('datasets/penguins_in_a_table.json', 'r') as json_file:
#    allEntries = json.load(json_file)
#    task = allEntries["task_prefix"]
#    allEntries = allEntries["examples"]
#    for entry in allEntries:
#        entry["input"] = task + '\n' + entry["input"]
#        json_list.append(entry)
jsonSamples = random.sample(range(len(json_list)), max_count)
print(jsonSamples)
#jsonSamples = [819, 1000, 699, 749, 580, 366, 1275, 203, 492, 773]
#jsonSamples = [269, 436, 836, 1291, 1283, 712, 391, 474, 1164, 783]
#jsonSamples = [13, 18]

for json_pos in tqdm(range(len(json_list)), desc="Going through reasoning questions"):
    result = json_list[json_pos]
    question = result["input"]
    target = result["target"]
    curResponse = {"question": question, "target": target}
    #question = mixed_questions[1]
    #print("Question: " + question)
    #print("Target: ", end="")
    #print(target)
    messages=[
        {"role": "system", "content": "You are a helpful assistant that accurately solves reasoning questions by acting as an agent who will choose libraries to use, perform Thoughts, where you will perform logical reasoning, and Actions, which perform functions."},
        {"role": "user", "content": libraryPrompt + '\n' + "Here is a question. " + '\n' + "Question: In 5 months, the date will be September 18th, 2016. What will be the date in 53 days?"},
        {"role": "assistant", "content": """Thought: To solve the question, I need to analyze the calendar dates and perform operations on them. Since the question involves working with calendar dates and performing calculations on them, I will use the Dates library.
Action: CHOOSE_LIBRARY[Dates]"""},
        {"role": "user", "content": allPrompts[1]},
        {"role": "assistant", "content": """Thought: If the date in 5 months will be September 18th, 2016, then today is the date 5 months before 9/18/2016. Thus, I need to find the date 5 months before 9/18/2016. 
Action: decMonth[9;18;2016;5]"""},
        {"role": "user", "content": "Observation: 4/18/2016"},
        {"role": "assistant", "content": """Thought: So today is 4/18/2016. We want to find the date in 53 days.
Action: incDay[4;18;2016;53]"""},
        {"role": "user", "content": "Observation: 6/10/2016"},
        {"role": "assistant", "content": """Thought: So in 53 days, the date will be 6/10/2016.
Action: FINISH[06/10/2016]"""},
        {"role": "user", "content": libraryPrompt + '\n' + """Question: Here is a table where the first line is a header and each subsequent line is a penguin:
name, age, height (cm), weight (kg) 
Louis, 7, 50, 11
Bernard, 5, 80, 13
Vincent, 9, 60, 11
Gwen, 8, 70, 15
For example: the age of Louis is 7, the weight of Gwen is 15 kg, the height of Bernard is 80 cm. 
We now add a penguin to the table:
James, 12, 90, 12
How many penguins are less than 8 years old?"""},
        {"role": "assistant", "content": """Thought: To solve the question, I need to create a table and perform operations on it. Since the question involves working with tables, I will use the Tables library.
Action: CHOOSE_LIBRARY[Tables]"""},
        {"role": "user", "content": allPrompts[2]},
        {"role": "assistant", "content": """Thought: I first need to create the table.
Action: CREATE_START[{"name": ["Louis", "Bernard", "Vincent", "Gwen"], "age": [7, 5, 9, 8], "height": [50, 80, 60, 70], "weight": [11, 13, 11, 15]}]CREATE_END
"""},
        {"role": "user", "content": "Observation: Created table"},
        {"role": "assistant", "content": """Thought: Now I need to add a new row for the new penguin
Action: ADD_START[{"name": "James", "age": 12, "height": 90, "weight": 12}]ADD_END"""},
        {"role": "user", "content": "Observation: Added row"},
        {"role": "assistant", "content": """Thought: Finally, we want to find the number of penguins whose age is less than 8.
Action: COUNT_START[age < 8]COUNT_END"""},
        {"role": "user", "content": "Observation: 2"},
        {"role": "assistant", "content": """Thought: So there are 2 penguins whose age is less than 8.
#Action: FINISH_START[2]FINISH_END"""},
        {"role": "user", "content": libraryPrompt + '\n' + "Here is a question. " + '\n' + "Question: " + question},
    ]
    #for message in messages:
    #    print(message["role"].upper())
    #    print(message["content"])
    #    print("")
    #break
    # interface.verbose = True
    finished = False
    curLibrary = None
    maxIters = 10
    numIters = 0
    while not finished:
        if numIters >= maxIters:
            print("Problem " + str(counter))
            print("Reached maximum number of iterations\n")
            curResponse["received"] = -987654321
            break
        numIters += 1
        gens = interface.generate_history(messages=messages, max_tokens=1024)
        output = gens[0]
        messages.append({"role": "assistant", "content": output})
        getLibrary = output[output.find("CHOOSE_LIBRARY[") + len("CHOOSE_LIBRARY["):output.find("]")]
        #print(output)
        if getLibrary == "Arithmetic":
            #print("Chosen: Arithmetic")
            curLibrary = 0
            finished = True
        elif getLibrary == "Dates":
            #print("Chosen: Dates")
            curLibrary = 1
            finished = True
        elif getLibrary == "Tables":
            #print("Chosen: Tables")
            curLibrary = 2
            finished = True
        else:
            messages.append({"role": "user", "content": "That was not a valid library. Valid libraries are: Arithmetic, Dates, and Tables."})
    if "received" in curResponse.keys():
        responses.append(curResponse)
        continue
    finished = False
    messages = []
    for message in allMessages[curLibrary]:
        messages.append(message)
    curPrompt = allPrompts[curLibrary]
    messages.append({"role": "user", "content": curPrompt + '\n' + "Question: " + question})
    
    curTable = None
    #print(question)
    #print("Target: " + str(target))
    maxIters = 40
    numIters = 0
    
    while not finished:
        if numIters >= maxIters:
            print("Problem " + str(counter))
            print("Reached maximum number of iterations\n")
            curResponse["received"] = -987654321
            break
        numIters += 1
        gens = interface.generate_history(messages=messages, max_tokens=1024)
        output = gens[0]
        #print(output)
        messages.append({"role": "assistant", "content": output})
        curAction = output[output.find("Action: ") + len("Action: "):]
        #print("Action: " + curAction)
        if "CHOOSE_LIBRARY" in output:
            getLibrary = output[output.find("CHOOSE_LIBRARY[") + len("CHOOSE_LIBRARY["):output.find("]")]
            if getLibrary == "Arithmetic":
                print("Chosen: Arithmetic")
                curLibrary = 0
            elif getLibrary == "Dates":
                print("Chosen: Dates")
                curLibrary = 1
            elif getLibrary == "Tables":
                print("Chosen: Tables")
                curLibrary = 2
            else:
                messages.append({"role": "user", "content": "That was not a valid library. Valid libraries are: Arithmetic, Dates, and Tables."})
            messages.append({"role": "user", "content": allPrompts[curLibrary]})
            continue
        if curLibrary == None:
            messages.append({"role": "user", "content": "You have not chosen a library yet."})
        elif curLibrary == 0:
            curAction = curAction.lower()
            expression = curAction[curAction.find('[') + 1:curAction.find(']')]
            expression = expression.replace(',', '') 
            #print(curAction)
            if "finish" in curAction:
                finished = True
                try:
                    curResponse["received"] = float(eval(expression))
                except:
                    print("#################################################")
                    print("Finish Error")
                    print("Question " + str(counter))
                    print(output)
                    curResponse["received"] = -987654321
            else:
                expression = expression.split(';')
                evaled = None
                try:
                    expression[0] = float(expression[0])
                    expression[1] = float(expression[1])
                except:
                    messages.append({"role": "user", "content": "Observation: That was not a valid action. Please try again. REMINDER: You cannot put variables as inputs to actions."})
                    continue
                if "add" in curAction:
                    evaled = str(expression[0] + expression[1])
                elif "sub" in curAction:
                    evaled = str(expression[0] - expression[1])
                elif "mult" in curAction:
                    evaled = str(expression[0] * expression[1])
                elif "div" in curAction:
                    evaled = str(expression[0] / expression[1])
                else:
                    evaled = "That was not a valid calculation. Please input a valid one."
                #messages.append({"role": "user", "content": "Observation: " + evaled + '\n' + reminderPrompt})
                messages.append({"role": "user", "content": "Observation: " + evaled})
        elif curLibrary == 1:
            curAction = curAction.lower()
            expression = curAction[curAction.find('[') + 1:curAction.find(']')]
            expression = expression.replace(',', '') # Commas are interpreted as tuples, so we remove them to avoid errors with eval
            #print(curAction)
            if "finish" in curAction:
                finished = True
                curResponse["received"] = expression
            else:
                expression = expression.split(';')
                curDay = None
                try:
                    curDay = datetime(int(expression[2]), int(expression[0]), int(expression[1]))
                    num = int(expression[3])
                except:
                    messages.append({"role": "user", "content": "Observation: That was not a valid action. Please input a valid one."})
                    continue
                if "incday" in curAction:
                    curDay = curDay + relativedelta(days=num)
                elif "incmonth" in curAction:
                    curDay = curDay + relativedelta(months=num)
                elif "incyear" in curAction:
                    curDay = curDay + relativedelta(years=num)
                elif "decday" in curAction:
                    curDay = curDay - relativedelta(days=num)
                elif "decmonth" in curAction:
                    curDay = curDay - relativedelta(months=num)
                elif "decyear" in curAction:
                    curDay = curDay - relativedelta(years=num)
                else:
                    evaled = "That was not a valid action. Please input a valid one."
                #messages.append({"role": "user", "content": "Observation: " + str(curDay.month) + '/' + str(curDay.day) + '/' + str(curDay.year) + '\n' + reminderPrompt})
                messages.append({"role": "user", "content": "Observation: " + str(curDay.month) + '/' + str(curDay.day) + '/' + str(curDay.year)})
        elif curLibrary == 2:
            #print(curAction)
            if "finish" in curAction.lower():
                finished = True
                expression = curAction[curAction.find("FINISH_START[") + len("FINISH_START["):curAction.find("]FINISH_END")]
                curResponse["received"] = expression
                #print(expression)
            else:
                evaled = None
                expression = None
                if "CREATE" in curAction:
                    expression = curAction[curAction.find("CREATE_START[") + len("CREATE_START["):curAction.find("]CREATE_END")]
                    try:
                        data = json.loads(expression)
                        curTable = pd.DataFrame(data)
                        evaled = "Created table"
                    except:
                        evaled = "That was not a valid input. Try again."
                    #print(curTable)
                elif "COUNT" in curAction:
                    expression = curAction[curAction.find("COUNT_START[") + len("COUNT_START["):curAction.find("]COUNT_END")]
                    try:
                        evaled = str(len(curTable.query(expression)))
                    except:
                        evaled = "That was not a valid query. It is possible that (1) The table does not exist yet. (2) The query is invalid."
                elif "FILTER" in curAction:
                    expression = curAction[curAction.find("FILTER_START[") + len("FILTER_START["):curAction.find("]FILTER_END")]
                    try:
                        curTable = curTable.query(expression)
                        curTable = curTable.reset_index(drop=True)
                        evaled = "Changed the table to:\n"
                        evaled += curTable.to_string()
                    except:
                        evaled = "That was not a valid query. It is possible that (1) The table does not exist yet. (2) The query is invalid."
                elif "GET_VALUE" in curAction:
                    expression = curAction[curAction.find("GET_VALUE_START[") + len("GET_VALUE_START["):curAction.find("]GET_VALUE_END")]
                    expression = expression.split(';')
                    expression[0] = int(expression[0])
                    try:
                        evaled = str(curTable[expression[1]][expression[0]])
                    except:
                        evaled = "That was not a valid input. It is possible that (1) The table does not exist yet. (2) The column name is wrong."
                elif "SORT" in curAction:
                    expression = curAction[curAction.find("SORT_START[") + len("SORT_START["):curAction.find("]SORT_END")]
                    try:
                        curTable = curTable.sort_values(by=expression)
                        curTable = curTable.reset_index(drop=True)
                        evaled = "Sorted table by " + expression
                    except:
                        evaled = "That was not a valid input. It is possible that (1) The table does not exist yet. (2) The column name is incorrect."
                elif "ADD" in curAction:
                    expression = curAction[curAction.find("ADD_START[") + len("ADD_START["):curAction.find("]ADD_END")]
                    try:
                        row = json.loads(expression)
                        new_row = pd.Series(row)
                        curTable = pd.concat([curTable, pd.DataFrame([new_row])], ignore_index=True)
                        evaled = "Added row."
                    except:
                        evaled = "That was not a valid input. Try again."
                else:
                    evaled = "That was not a valid action. Please input a valid one."
                #print("Expression: ", end = "")
                #print(expression)
                
                #messages.append({"role": "user", "content": "Observation: " + evaled + '\n' + reminderPrompt})
                messages.append({"role": "user", "content": "Observation: " + evaled + '\n'})
                
                #print("Evaluation: ", end="")
                #print(evaled)
                #print(messages[-1]["content"])
    curResponse["messages"] = messages
    responses.append(curResponse)
    counter += 1
numCorrect = 0

for response in responses:
    #print("Received: " + str(response["received"]))
    #print("Target: ", end="")
    #print(response["target"])
    #print("")
    #for message in response["messages"]:
    #    print(message["content"])
    try:
        if abs(response["received"] - response["target"]) < 1e-6:
            numCorrect += 1
    except:
        pass
    #if response["received"] not in response["target"].keys():
    #        continue
    #if response["target"][response["received"]] == 1:
    #    numCorrect += 1
    #if response["received"] == response["target"]:
    #    numCorrect += 1
print("Number correct = " + str(numCorrect))

[828, 855, 675, 766, 37]


Going through reasoning questions:   0%|          | 0/1000 [00:00<?, ?it/s]

#################################################
Finish Error
Question 27
Thought: Similar to the crackers, he cannot give a fraction of a cake either. Therefore, he can only give whole numbers. Since he cannot give 1 cracker to each friend, he also cannot give 1 cake to each friend. 

Thought: I need to determine the maximum number of whole items he can give to each friend. Since he has 13 crackers, he can give 0 crackers and 0 cakes to each friend. 

Action: FINISH_START[0;0]FINISH_END
#################################################
Finish Error
Question 135
Thought: Since the result is negative, it indicates that the farmer cannot pick 172 potatoes because he only has 77. Therefore, he will have 0 potatoes left after picking all of them. Now, I need to find out how many tomatoes he has left, which remains unchanged at 175. 
Action: FINISH_START[175;0]FINISH_END
#################################################
Finish Error
Question 227
Thought: Each friend can receive 2 crackers.

In [21]:
numCorrect = 0
curCount = 0
for response in responses:
    #print("Received: " + str(response["received"]))
    #print("Target: ", end="")
    #print(response["target"])
    #print("")
    #for message in response["messages"]:
    #    print(message["content"])
    try:
        if abs(response["received"] - response["target"]) < 1e-6:
            numCorrect += 1
    except:
        pass
    #if response["received"] not in response["target"].keys():
    #        continue
    #if response["target"][response["received"]] == 1:
    #    numCorrect += 1
    #if response["received"] == response["target"]:
    #    numCorrect += 1
print("Number correct = " + str(numCorrect))
print("Total = " + str(len(responses)))
#with open('gsm-results-03-next215-gpt-4o-mini.json', 'w') as f:
#    json.dump(responses, f)
print(numCorrect)

Number correct = 195
Total = 215
195


In [72]:

curPos = 0
print("Target:", end=' ')
print(responses[curPos]["target"])
print("")
print(type(responses[curPos]["target"]))
print(type(responses[curPos]["received"]))
for message in responses[curPos]["messages"]:
    print(message["role"].upper())
    print(message["content"])
    print("")

Target: 738

<class 'int'>
<class 'int'>
SYSTEM
You are a helpful assistant that accurately solves reasoning questions by acting as an agent who will choose libraries to use, perform Thoughts, where you will perform logical reasoning, and Actions, which perform functions.

USER
There are 3 kinds of libraries you can use:
(1) Arithmetic - This contains functions that allow you to perform arithmetic operations
(2) Dates - This contains functions that allow you to perform operations on calendar dates
(3) Tables - This contains functions that allow you to perform operations on tables.
To choose a library to use, write CHOOSE_LIBRARY[library_name]. The list of functions in the library will be provided to you. For example, if you want to use Dates, write CHOOSE_LIBRARY[Dates].

Here is a question. 
Question: In 5 months, the date will be September 18th, 2016. What will be the date in 53 days?

ASSISTANT
Thought: To solve the question, I need to analyze the calendar dates and perform operatio