In [1]:
%load_ext autoreload
%autoreload 2

In [19]:
# File paths
import os
dataset_root = '../MATH/train/'

train_sample_file = os.path.join(dataset_root, 'train_gr20_samples.json')
output_file = os.path.join(dataset_root, 'train_gr20_cot_qa.json')
train_sample_cot_file = os.path.join(
    dataset_root, 'train_gr20_samples_cot.json')
grouped_training_file = os.path.join(
    dataset_root, 'train_gr20.json')


In [20]:
import json
import os
from openai import OpenAI
from tqdm import tqdm  # Optional: For a progress bar

# Load train sample data
with open(train_sample_file, 'r') as f:
    train_sample = json.load(f)
train_sample['0']

[{'809': {'conversations': [{'from': 'human',
     'value': "An isosceles trapezoid has legs of length 30 cm each, two diagonals of length 40 cm each and the longer base is 50 cm. What is the trapezoid's area in sq cm?"},
    {'from': 'gpt',
     'value': 'We can pick a diagonal and a leg of the trapezoid such that, along with the longer base, these lines form a triangle with sides of length 30, 40, and 50. This is a Pythagorean triple, so the triangle is a right triangle. It follows that the altitude to the longer base of the trapezoid is $30\\cdot 40/50 = 24$. This altitude is the same length as the height of the trapezoid.\n\nWe now look at the right triangle formed by this altitude, the adjacent leg of the trapezoid, and part of the longer base. These three sides form a right triangle, with hypotenuse of 30 and one leg (the altitude) of length 24. It follows that the other leg has length 18.\n\nBecause this is an isosceles trapezoid, we can now calculate the shorter base to have le

In [4]:
# Enhanced data storage
enhanced_data = {}

client = OpenAI(
    api_key=os.environ.get("OPENAI_API_KEY"),
    base_url=os.environ.get("OPENAI_API_BASE")
)

# Iterate through each group
for group_number, group_items in tqdm(train_sample.items(), desc="Processing groups"):
    enhanced_group = []
    for item in group_items:
        item = next(iter(item.values()))
        # Extract question and answer from the conversations
        question = item['conversations'][0]['value']
        answer = item['conversations'][1]['value']

        # Generate enhanced CoT explanation using the OpenAI API
        try:
            response = client.chat.completions.create(
                model="gpt-4o-mini",
                messages=[
                    {"role": "system", "content": "You are a helpful assistant specializing in generating well structured explanations for educational content."},
                    {
                        "role": "user",
                        "content": f"Generate a detailed chain of thought explanation for the answer, do not include the question itself. In other words, you need to write a step by step explanation that enhances the original answer. Also, you need to follow the original answer's format, that is 1. Use latex format when encountering math formula. \n2. Do not add formatting to your answer like \\textbf, \\text, that is not neccessary for the deduction. \n3. Your final answer is wrapped inside \\boxed for future extraction. For example,  To two decimal places, $\\pi=\\boxed{3.14}$ \n4. For math formula enviroment, only use single $ sign, not double $$ signs. That is only use inline math environment!\n5. The questions are simple, so do not make the cot answer too long. \n\n Here is the original answer together with the question, you can see that it follows the format we just talked about, you need to follow the format. Question: {question}\nOrignal answer:{answer}\n\nYour step by step explanation:"
                    }
                ]
            )
            cot_answer = response.choices[0].message.content

            # Create a new item with the CoT answer
            enhanced_item = item.copy()
            enhanced_item['conversations'].append({
                'from': 'cot',
                'value': cot_answer
            })
            enhanced_group.append(enhanced_item)

        except Exception as e:
            print(f"Error processing group {group_number}, item: {question}. Error: {e}")
            continue

    # Store enhanced group in the dictionary
    enhanced_data[group_number] = enhanced_group

    # Save the enhanced data to a new file
    with open(train_sample_cot_file, 'w') as f:
        json.dump(enhanced_data, f, indent=2, ensure_ascii=False)

Processing groups: 100%|██████████| 20/20 [18:50<00:00, 56.53s/it]


In [7]:
enhanced_data['0'][0]

{'conversations': [{'from': 'human',
   'value': "An isosceles trapezoid has legs of length 30 cm each, two diagonals of length 40 cm each and the longer base is 50 cm. What is the trapezoid's area in sq cm?"},
  {'from': 'gpt',
   'value': 'We can pick a diagonal and a leg of the trapezoid such that, along with the longer base, these lines form a triangle with sides of length 30, 40, and 50. This is a Pythagorean triple, so the triangle is a right triangle. It follows that the altitude to the longer base of the trapezoid is $30\\cdot 40/50 = 24$. This altitude is the same length as the height of the trapezoid.\n\nWe now look at the right triangle formed by this altitude, the adjacent leg of the trapezoid, and part of the longer base. These three sides form a right triangle, with hypotenuse of 30 and one leg (the altitude) of length 24. It follows that the other leg has length 18.\n\nBecause this is an isosceles trapezoid, we can now calculate the shorter base to have length $50 - 2\\c

In [21]:
import json
import random

# Load grouped training data
grouped_training = []
with open(grouped_training_file, 'r') as f:
    grouped_training = json.load(f)
grouped_training[0]

{'0': {'conversations': [{'from': 'human',
    'value': 'A rectangle having integer length and width has a perimeter of 100 units. What is the number of square units in the least possible area?'},
   {'from': 'gpt',
    'value': 'A rectangle with fixed perimeter has minimal area when one dimension is as long as possible and the other is as short as possible. To see this, let $x$ be the shorter dimension and $y$ the area of the rectangle, and note that $y=x(50-x)$. The graph of $y=x(50-x)$ is a down-turned parabola with vertex at $(25,625)$, and thus is as small as possible when $x$ is as small as possible. Since $x$ is an integer, its minimum value is 1. Thus the relevant rectangle with minimum area is 1 by 49.  Its area is $49\\cdot 1=\\boxed{49}$ square units.'}],
  'system': 'You are a helpful assistant skilled in solving mathematical problems. Provide step-by-step explanations for each solution and ensure clarity in reasoning.',
  'tools': {'calculator': {'description': 'A basic ca

In [22]:
# Load sampled training data
with open(train_sample_cot_file, 'r') as f:
    train_sample = json.load(f)
train_sample['0']


[{'conversations': [{'from': 'human',
    'value': "An isosceles trapezoid has legs of length 30 cm each, two diagonals of length 40 cm each and the longer base is 50 cm. What is the trapezoid's area in sq cm?"},
   {'from': 'gpt',
    'value': 'We can pick a diagonal and a leg of the trapezoid such that, along with the longer base, these lines form a triangle with sides of length 30, 40, and 50. This is a Pythagorean triple, so the triangle is a right triangle. It follows that the altitude to the longer base of the trapezoid is $30\\cdot 40/50 = 24$. This altitude is the same length as the height of the trapezoid.\n\nWe now look at the right triangle formed by this altitude, the adjacent leg of the trapezoid, and part of the longer base. These three sides form a right triangle, with hypotenuse of 30 and one leg (the altitude) of length 24. It follows that the other leg has length 18.\n\nBecause this is an isosceles trapezoid, we can now calculate the shorter base to have length $50 - 

In [14]:
# Ensure train_sample has exactly 20 lists (one for each group)
assert len(train_sample) == 20, "Train sample must contain 20 groups."


for item in grouped_training:
    group_number = next(iter(item.values()))['group_number']
    # Get a random sample from the same group
    cot_prompt = random.choice(train_sample[str(group_number)])
    # Construct the new JSON object
    item['cot_prompt'] = cot_prompt

with open(output_file, 'w') as f:
    json.dump(grouped_training, f, indent=2, ensure_ascii=False)

In [18]:
grouped_training[0]['cot_prompt']

{'conversations': [{'from': 'human',
   'value': 'What is the area, in square inches, of a right triangle with a 24-inch leg and a 25-inch hypotenuse?'},
  {'from': 'gpt',
   'value': 'The other leg is $\\sqrt{25^2-24^2}=\\sqrt{625-576}=\\sqrt{49}=7$. The area is $\\frac12\\cdot24\\cdot7=12\\cdot7=\\boxed{84}$ square inches.'},
  {'from': 'cot',
   'value': 'To find the area of the right triangle, we first need to identify its dimensions. We are given one leg, which is 24 inches, and the hypotenuse, which is 25 inches. \n\n1. We will use the Pythagorean theorem, which states that in a right triangle, the square of the length of the hypotenuse ($c$) is equal to the sum of the squares of the lengths of the other two legs ($a$ and $b$). This relationship can be expressed as:\n   $$ c^2 = a^2 + b^2 $$\n\n2. In our case, we can assign the values as follows:\n   - One leg $a = 24$ inches\n   - Hypotenuse $c = 25$ inches\n   - We need to find the length of the other leg $b$.\n\n3. Substitutin