In [1]:
import pandas as pd
import numpy as np
import os
import re

# Processing utilities

In [44]:
B_PYTHON, E_PYTHON = '[PYTHON]', '[/PYTHON]'
B_CPP, E_CPP = '[CPP]', '[/CPP]'
B_INST, E_INST = '[INST]', '[/INST]'

In [56]:
cpp_func_match_reg_exp = "\\b[A-Za-z_][A-Za-z0-9_]*\s+[A-Za-z_][A-Za-z0-9_]*\s*\([^;{]*\)\s*{"
python_func_match_reg_exp = "def\s+[\w]+\s*\(.*?\):"

In [163]:
def end_overlap(a, b):
    for i in range(0, len(a)):
        if b.startswith(a[i:]):
            return i
    return len(a)

def remove_prefix(s, prefix):
    prefix_end_idx = s.index(prefix) + len(prefix)
    return s[prefix_end_idx:]

def clean_code(code):
    # cleans lines like these where appending of code context and response creates - def CalculateDifference(arr, n):def calculate_difference(arr, n):
    lines = code.split('\n')
    cleaned_code = []
    for line in lines:
        if line.count('def ') == 2:
            code_context, response = line.split(':', 1)
            response = response.strip()
            cleaned_code.append(response)
        else:
            cleaned_code.append(line)

    return '\n'.join(cleaned_code)

def clean_duplicate_func_signature(code, reg_exp):
    """
    for cpp files and python files - clean lines like the following which have duplicate function signatures - 

        int countSol ( int coeff [ ] , int start , int end , int rhs ) {
        int countSol(int coeff[], int start, int end, int rhs) {
    
    """
    lines = code.split('\n')
    cleaned_code = []
    for i, line in enumerate(lines):
        match = re.search(reg_exp, line)
        if match and i < len(lines)-1:
            second_match = re.search(reg_exp, lines[i+1])
            if not second_match:
                cleaned_code.append(line)
        else:
            cleaned_code.append(line)

    return '\n'.join(cleaned_code)

def remove_comments_at_beginning_python(code):
    lines = code.split('\n')

    cleaned_code = []
    found_code = False
    for line in lines:
        if not found_code and line.strip().startswith('#'):
            continue
        else:
            found_code = True
            cleaned_code.append(line)
    return '\n'.join(cleaned_code)

def remove_blank_lines(code):
    lines = code.split('\n')

    cleaned_code = []
    for line in lines:
        if line.strip() != '':
            cleaned_code.append(line)

    return '\n'.join(cleaned_code)

def extract_code_python(prompt, response, code_context):
    trimmed_response = remove_prefix(response, prompt)
    # print('stage1: ', trimmed_response)
    if trimmed_response.startswith(B_PYTHON):
        trimmed_response = remove_prefix(trimmed_response, B_PYTHON)

    # print('stage2: ', trimmed_response)
    try:
        suffix_idx = trimmed_response.index(E_PYTHON)
        trimmed_response = trimmed_response[:suffix_idx]
    except ValueError as ve:
          pass
    # print('stage3: ', trimmed_response)
    trimmed_response = trimmed_response.lstrip()
    code_context = code_context.strip()
    # print('stage4 code_context: ', code_context)

    # remove any comments at the starting of the response
    trimmed_response = remove_comments_at_beginning_python(trimmed_response)
    
    overlap_idx = end_overlap(code_context, trimmed_response)

    cleaned_response = trimmed_response
    if len(code_context[:overlap_idx]) > 0:
        cleaned_response = code_context[:overlap_idx] + '\n' + cleaned_response

    # print('stage5: ', cleaned_response)
    data = clean_duplicate_func_signature(clean_code(cleaned_response), python_func_match_reg_exp)

    # print('stage6: ', data)
    return data

def extract_code_cpp(prompt, response, code_context):
    trimmed_response = remove_prefix(response, prompt)
    if trimmed_response.startswith(B_CPP):
        trimmed_response = remove_prefix(trimmed_response, B_CPP)
    try:
        suffix_idx = trimmed_response.index(E_CPP)
        trimmed_response = trimmed_response[:suffix_idx]
    except ValueError as ve:
          pass

    try:
        suffix_idx = trimmed_response.index(E_INST)
        trimmed_response = trimmed_response[:suffix_idx]
    except ValueError as ve:
          pass

    trimmed_response = trimmed_response.strip()
    code_context = code_context.strip()
    overlap_idx = end_overlap(code_context, trimmed_response)
    cleaned_response = code_context[:overlap_idx] + '\n' + trimmed_response
    return clean_duplicate_func_signature(clean_code(cleaned_response), cpp_func_match_reg_exp)

def save_file(path, code):
    with open(path, 'w', encoding='utf-8') as file:
        file.write(code)

def save_to_code_file(directory, id, code, ext, pss=1):
    save_file(f"{directory}/question-{id}-pass-{pss}.{ext}", code)

# Python processing

In [164]:
python_responses_df = pd.read_csv('./python/train_response_with_feedback.csv', index_col=0)
python_responses_df.head()

Unnamed: 0,id,text,code,question,functions,code_context,context,llama_prompt,response,response_code,report,feedback,llama_prompt_with_feedback,response_feedback
0,0,Maximum Prefix Sum possible by merging two giv...,"def maxPresum(a, b):\n X = max(a[0], 0)\n ...",Maximum Prefix Sum possible by merging two giv...,"['def maxPresum(a, b):']","def maxPresum(a, b):","def maxPresum(a, b):",<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"def maxPresum(a, b):\n max_sum = 0\n for...",{'./files/question-0-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...
1,1,Check if a number can be represented as sum of...,import math\n\n\ndef sumOfTwoCubes(n):\n lo...,Check if a number can be represented as sum of...,['def sumOfTwoCubes(n):'],import math\n\n\ndef sumOfTwoCubes(n):,import math\n\n\ndef sumOfTwoCubes(n):,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,import math\n\n\ndef sumOfTwoCubes(n):\n fo...,{'./files/question-1-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...
2,2,Generate an N | Python3 program for the above ...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,Generate an N.,"['def sieveOfPrimes():', 'def getArray(arr, N):']",sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,{'./files/question-2-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...
3,3,Nth natural number after removing all numbers ...,def findNthNumber(N):\n result = 0\n p =...,Nth natural number after removing all numbers ...,['def findNthNumber(N):'],def findNthNumber(N):,def findNthNumber(N):,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,def findNthNumber(N):\n nth_number = 1\n ...,{'./files/question-3-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...
4,4,Check if an integer is rotation of another giv...,"import math\n\n\ndef check(A, B):\n if (A =...",Check if an integer is rotation of another giv...,"['def check(A, B):']","import math\n\n\ndef check(A, B):","import math\n\n\ndef check(A, B):",<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"import math\n\n\ndef check(A, B):\n if A ==...",{'./files/question-4-pass-1.py': [{'code': 'E2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...


In [165]:
print(python_responses_df.loc[298]['response_feedback'])

<s>[INST]<<SYS>>
Provide answers in Python. Wrap around code in [PYTHON] and [/PYTHON]. Don't write any comments in generated code.
<</SYS>>

# Count Numbers with N digits which consists of odd number of 0 's.
def countNumbers(N):[/INST]def count_numbers(n):
    count = 0
    for i in range(1, 10**n):
        if str(i).count("0") % 2 == 1:
            count += 1
    return count</s><s>[INST]Regenerate the entire code after fixing the following errors: 
1. no newline at end of file at line no. 6 around code - `    return`
Wrap around code in [PYTHON] and [/PYTHON]. 
# Count Numbers with N digits which consists of odd number of 0 's.
def countNumbers(N):[/INST]
[PYTHON]
def count_numbers(n):
    count = 0
    for i in range(1, 10**n):
        if str(i).count("0") % 2 == 1:
            count += 1
    return count
[/PYTHON]



In [166]:
def create_directory_if_not_exists(directory):
    if not os.path.exists(directory):
        os.makedirs(directory)

In [167]:
PYTHON_SAVE_PATH = './python/files_with_feedback'
create_directory_if_not_exists(PYTHON_SAVE_PATH)

In [168]:
print(extract_code_python(python_responses_df.loc[298]['llama_prompt_with_feedback'], python_responses_df.loc[298]['response_feedback'], python_responses_df.loc[298]['code_context']))

def count_numbers(n):
    count = 0
    for i in range(1, 10**n):
        if str(i).count("0") % 2 == 1:
            count += 1
    return count



In [169]:
python_responses_df['response_code_post_feedback'] = python_responses_df.apply(lambda row: extract_code_python(row.llama_prompt_with_feedback, row.response_feedback, row.code_context), axis=1)
python_responses_df.head()

Unnamed: 0,id,text,code,question,functions,code_context,context,llama_prompt,response,response_code,report,feedback,llama_prompt_with_feedback,response_feedback,response_code_post_feedback
0,0,Maximum Prefix Sum possible by merging two giv...,"def maxPresum(a, b):\n X = max(a[0], 0)\n ...",Maximum Prefix Sum possible by merging two giv...,"['def maxPresum(a, b):']","def maxPresum(a, b):","def maxPresum(a, b):",<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"def maxPresum(a, b):\n max_sum = 0\n for...",{'./files/question-0-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"def maxPresum(a, b):\n max_sum = 0\n for..."
1,1,Check if a number can be represented as sum of...,import math\n\n\ndef sumOfTwoCubes(n):\n lo...,Check if a number can be represented as sum of...,['def sumOfTwoCubes(n):'],import math\n\n\ndef sumOfTwoCubes(n):,import math\n\n\ndef sumOfTwoCubes(n):,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,import math\n\n\ndef sumOfTwoCubes(n):\n fo...,{'./files/question-1-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,import math\n\n\ndef sumOfTwoCubes(n):\nimport...
2,2,Generate an N | Python3 program for the above ...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,Generate an N.,"['def sieveOfPrimes():', 'def getArray(arr, N):']",sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...,{'./files/question-2-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,sieve = [1] * (1000000 + 1)\n\n\ndef sieveOfPr...
3,3,Nth natural number after removing all numbers ...,def findNthNumber(N):\n result = 0\n p =...,Nth natural number after removing all numbers ...,['def findNthNumber(N):'],def findNthNumber(N):,def findNthNumber(N):,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,def findNthNumber(N):\n nth_number = 1\n ...,{'./files/question-3-pass-1.py': [{'code': 'W2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,def findNthNumber(N):\n nth_number = 1\n ...
4,4,Check if an integer is rotation of another giv...,"import math\n\n\ndef check(A, B):\n if (A =...",Check if an integer is rotation of another giv...,"['def check(A, B):']","import math\n\n\ndef check(A, B):","import math\n\n\ndef check(A, B):",<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"import math\n\n\ndef check(A, B):\n if A ==...",{'./files/question-4-pass-1.py': [{'code': 'E2...,Regenerate the entire code after fixing the fo...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,<s>[INST]<<SYS>>\nProvide answers in Python. W...,"import math\n\n\n\ndef check(A, B):\n if A ..."


In [170]:
python_responses_df.to_csv('./python/train_response_with_feedback_processed.csv')

In [171]:
python_responses_df.apply(lambda row: save_to_code_file(PYTHON_SAVE_PATH, row.id, row.response_code_post_feedback, "py"), axis=1)

0      None
1      None
2      None
3      None
4      None
       ... 
295    None
296    None
297    None
298    None
299    None
Length: 300, dtype: object

In [152]:
# random_samples = python_responses_df['response_code'].sample(20).values.tolist()

# for i, code in enumerate(random_samples):
#     print(i)
#     print(code, '\n')

In [38]:
# # def calculate_difference(arr, n)
# python_responses_df[python_responses_df['response_code'].str.contains('from math import sqrt')]

In [39]:
# print(python_responses_df.loc[288]['response'])

# C++ processing

In [131]:
cpp_responses_df = pd.read_csv('./cpp/train_response.csv')
cpp_responses_df.head()

Unnamed: 0,id,text,code,question,code_cleaned,functions,code_context,context,llama_prompt,response
0,0,Maximum Prefix Sum possible by merging two giv...,#include <bits/stdc++.h> NEW_LINE using namesp...,Maximum Prefix Sum possible by merging two giv...,#include <bits/stdc++.h> \n using namespace st...,"['int maxPresum ( vector < int > a , vector < ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...
1,1,Check if a number can be represented as sum of...,#include <bits/stdc++.h> NEW_LINE using namesp...,Check if a number can be represented as sum of...,#include <bits/stdc++.h> \n using namespace st...,"['bool sumOfTwoCubes ( int n ) {', 'int main (...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...
2,3,Nth natural number after removing all numbers ...,#include <bits/stdc++.h> NEW_LINE using namesp...,Nth natural number after removing all numbers ...,#include <bits/stdc++.h> \n using namespace st...,"['long findNthNumber ( long long N ) {', 'int ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...
3,4,Check if an integer is rotation of another giv...,#include <bits/stdc++.h> NEW_LINE using namesp...,Check if an integer is rotation of another giv...,#include <bits/stdc++.h> \n using namespace st...,"['int check ( int A , int B ) {', 'int main ( ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...
4,5,Count of quadruples with product of a pair equ...,#include <bits/stdc++.h> NEW_LINE using namesp...,Count of quadruples with product of a pair equ...,#include <bits/stdc++.h> \n using namespace st...,"['void sameProductQuadruples ( int nums [ ] , ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...


In [132]:
len(cpp_responses_df)

300

In [136]:
CPP_SAVE_PATH = './cpp/files'
create_directory_if_not_exists(CPP_SAVE_PATH)

In [133]:
cpp_responses_df['response_code'] = cpp_responses_df.apply(lambda row: extract_code_cpp(row.llama_prompt, row.response, row.code_context), axis=1)
cpp_responses_df.head()

Unnamed: 0,id,text,code,question,code_cleaned,functions,code_context,context,llama_prompt,response,response_code
0,0,Maximum Prefix Sum possible by merging two giv...,#include <bits/stdc++.h> NEW_LINE using namesp...,Maximum Prefix Sum possible by merging two giv...,#include <bits/stdc++.h> \n using namespace st...,"['int maxPresum ( vector < int > a , vector < ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,#include <bits/stdc++.h> \n using namespace st...
1,1,Check if a number can be represented as sum of...,#include <bits/stdc++.h> NEW_LINE using namesp...,Check if a number can be represented as sum of...,#include <bits/stdc++.h> \n using namespace st...,"['bool sumOfTwoCubes ( int n ) {', 'int main (...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,#include <bits/stdc++.h> \n using namespace st...
2,3,Nth natural number after removing all numbers ...,#include <bits/stdc++.h> NEW_LINE using namesp...,Nth natural number after removing all numbers ...,#include <bits/stdc++.h> \n using namespace st...,"['long findNthNumber ( long long N ) {', 'int ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,#include <bits/stdc++.h> \n using namespace st...
3,4,Check if an integer is rotation of another giv...,#include <bits/stdc++.h> NEW_LINE using namesp...,Check if an integer is rotation of another giv...,#include <bits/stdc++.h> \n using namespace st...,"['int check ( int A , int B ) {', 'int main ( ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,#include <bits/stdc++.h> \n using namespace st...
4,5,Count of quadruples with product of a pair equ...,#include <bits/stdc++.h> NEW_LINE using namesp...,Count of quadruples with product of a pair equ...,#include <bits/stdc++.h> \n using namespace st...,"['void sameProductQuadruples ( int nums [ ] , ...",#include <bits/stdc++.h> \n using namespace st...,#include <bits/stdc++.h> \n using namespace st...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,<s>[INST]<<SYS>>\nProvide answers in CPP. Wrap...,#include <bits/stdc++.h> \n using namespace st...


In [134]:
# random_samples = cpp_responses_df['response_code'].sample(20).values.tolist()

# for i, code in enumerate(random_samples):
#     print(i)
#     print(code, '\n')

In [135]:
cpp_responses_df.to_csv('./cpp/train_response_processed.csv')

In [150]:
cpp_responses_df.apply(lambda row: save_to_code_file(CPP_SAVE_PATH, row.id, row.response_code, "cpp"), axis=1)

0      None
1      None
2      None
3      None
4      None
       ... 
295    None
296    None
297    None
298    None
299    None
Length: 300, dtype: object