In [2]:
import re
import pandas as pd

data = pd.read_json("sanitized-mbpp.json")

def extract_function_info(row):
    match = re.search(r'def\s+([\w_]+)\((.*?)\):([\s\S]*?)\n', row.code)
    if match:
        function_name = match.group(1)
        arguments = match.group(2).split(',')
        return "Given function arguments " + ','.join(arguments)

data['context'] = data.apply(extract_function_info, axis=1)

system_message = """You are an text to PYTHON function translator. Users will ask you questions in English and you will generate a PYTHON function based on the provided CONTEXT.
CONTEXT:
{context}"""

def create_conversation(sample):
    return {
        "messages": [
            {"role": "system", "content": system_message.format(context=sample["context"])},
            {"role": "user", "content": sample["prompt"]},
            {"role": "assistant", "content": sample["code"]}
        ]
    }

dataset = data.apply(create_conversation, axis=1)
dataset = dataset.sample(frac=1, random_state=42).reset_index(drop=True)
train_idx = int(len(dataset)*0.8)
train = dataset[:train_idx]
test = dataset[train_idx:]

train.to_json("train_dataset.json", orient="records")
test.to_json("test_dataset.json", orient="records")

In [3]:
train.iloc[0]

{'messages': [{'role': 'system',
   'content': 'You are an text to PYTHON function translator. Users will ask you questions in English and you will generate a PYTHON function based on the provided CONTEXT.\nCONTEXT:\nGiven function arguments arr'},
  {'role': 'user',
   'content': 'Write a python function to count the number of rotations required to generate a sorted array. https://www.geeksforgeeks.org/count-of-rotations-required-to-generate-a-sorted-array/'},
  {'role': 'assistant',
   'content': 'def count_rotation(arr):   \n    for i in range (1,len(arr)): \n        if (arr[i] < arr[i - 1]): \n            return i  \n    return 0'}]}

In [4]:
test.iloc[0]

{'messages': [{'role': 'system',
   'content': 'You are an text to PYTHON function translator. Users will ask you questions in English and you will generate a PYTHON function based on the provided CONTEXT.\nCONTEXT:\nGiven function arguments n'},
  {'role': 'user',
   'content': 'Write a function to calculate the sum of perrin numbers.'},
  {'role': 'assistant',
   'content': 'def cal_sum(n): \n\ta = 3\n\tb = 0\n\tc = 2\n\tif (n == 0): \n\t\treturn 3\n\tif (n == 1): \n\t\treturn 3\n\tif (n == 2): \n\t\treturn 5\n\tsum = 5\n\twhile (n > 2): \n\t\td = a + b \n\t\tsum = sum + d \n\t\ta = b \n\t\tb = c \n\t\tc = d \n\t\tn = n-1\n\treturn sum'}]}