In [1]:
import json
import os

bird_dev_file = r'C:\Users\SatyaSaurabh_Mishra\Downloads\workspace\Paper for ICML\Text2sql\exploring-text-to-sql\data\sub_sampled_bird_dev_set.json'

with open(bird_dev_file, 'r') as file:
    bird_dev_set = json.load(file)

In [2]:
bird_dev_set[3]

{'question_id': 35,
 'db_id': 'california_schools',
 'question': "What is the administrator's email address of the chartered school with the fewest students enrolled in grades 1 through 12?",
 'evidence': 'Charted school means `Charter School (Y/N)` = 1 in the table frpm; Students enrolled in grades 1 through 12 refers to `Enrollment (K-12)`',
 'SQL': 'SELECT T2.AdmEmail1 FROM frpm AS T1 INNER JOIN schools AS T2 ON T1.CDSCode = T2.CDSCode WHERE T1.`Charter School (Y/N)` = 1 ORDER BY T1.`Enrollment (K-12)` ASC LIMIT 1',
 'difficulty': 'moderate'}

In [3]:
import pandas as pd
from collections import defaultdict

col_hash = defaultdict(list)

database = "california_schools"
ir = ["`schools`.`CDSCode`", "`schools`.`AdmEmail1`","`frpm`.`CDSCode`","`frpm`.`Charter School (Y/N)`","`frpm`.`Enrollment (K-12)`"]


for col in ir:
    table = col.split('.')[0].strip('`')
    col = col.split('.')[1].strip('`')
    col_hash[table].append(col)
    
print(col_hash)

test_question = bird_dev_set[3]['question']
test_hint = bird_dev_set[3]['evidence']


defaultdict(<class 'list'>, {'schools': ['CDSCode', 'AdmEmail1'], 'frpm': ['CDSCode', 'Charter School (Y/N)', 'Enrollment (K-12)']})


# Query rephrasing

In [4]:
# creating instruction string
instruction_str = """<s>[INST] <<SYS>>You are expert in rephrasing the question. Given the question, rephrase it into a new question by 
identifying and incorporating relevant information from the given columns by breaking the process into step by step such 
that the final rephrased question expliciitly mentions the columns needed in SQL to answer the questions.

** Please Note ** :
    - Provide the rephrased questions maintaining the context and intent of the original question. 
    - Use (refer - table.column name) format to add details in final rephrased query.
    - DO NOT USE ANY INFORMATION FROM THE EXAMPLE IN YOUR ANSWER. IT IS MEANT ONLY FOR YOUR UNDERSTANDING.

** Inputs ** :
Original Question: This is natural language question that need to rephrased.
Hint: This is hint for the original question for your help.
Required Columns: This is a dictionary in which the keys are table and the values are list of columns. 

** Output **:
The output should be a rephrased version of the question, rewritten to include the columns names in question,
so it becomes clear which SQL columns are being referred to. 
<</SYS>>
"""

# creating question string
question_str = f"""
Question: {test_question}
Hint: {test_hint}
Columns:{col_hash}
"""

In [5]:
# creating one shot example string (create more examples later)
one_shot_example = """
<example>
**Here is the EXAMPLE for your Understanding:**

**Original question**: In which city can you find the school in the state of California with the lowest latitude coordinates and what is its lowest grade? Indicate the school name.
**Hint**: State of California refers to state = 'CA'
**Columns**: defaultdict(<class 'list'>, {'schools': ['City', 'CDSCode', 'State', 'Latitude'], 'frpm': ['Low Grade', 'School Name', 'CDSCode']})

Steps to rephrase the Query:

* **Step 1:** Identify the Key Components of the Question
* **Analysis:**
-"City" where the school is located
-"Lowest latitude coordinates" of the school
-"State" which is California ('CA') from the hint
-"Lowest grade" of the school
-"School name"
* **Step 2:** Map the Identified Components to the Columns
* **Analysis:**
-"City" -> City (from 'schools' table)
-"Lowest latitude coordinates" -> Latitude (from 'schools' table)
-"State" (California) -> State (from 'schools' table) with value 'CA'
-"Lowest grade" -> Low Grade (from 'frpm' table)
-"School name" -> School Name (from 'frpm' table)
-Common identifier to link both tables -> CDSCode
* **Step 3:** Construct the rephrased Query:
    -Extract the city and school name by identifying the school with the lowest latitude in California.
    -Ensure the query incorporates both "Low Grade" and "School Name" columns.
    -Utilize the CDSCode to join the tables.

**Final rephrased Question": In which city (refer - schools.City) can you find the school in the state of California (refer - schools.State = 'CA') with the lowest latitude (refer - schools.Latitude) coordinates, and what is its lowest grade (refer - frpm.Low Grade) Indicate the school name (refer - frpm.School Name).
</example>
"""

In [6]:
# creating full prompt
query_rephrase_prompt = f"""{instruction_str}
**************************
{one_shot_example}
**************************
{question_str}
**************************
ANSWER: 
[/INST]</s>
"""

In [7]:
print(query_rephrase_prompt)

<s>[INST] <<SYS>>You are expert in rephrasing the question. Given the question, rephrase it into a new question by 
identifying and incorporating relevant information from the given columns by breaking the process into step by step such 
that the final rephrased question expliciitly mentions the columns needed in SQL to answer the questions.

** Please Note ** :
    - Provide the rephrased questions maintaining the context and intent of the original question. 
    - Use (refer - table.column name) format to add details in final rephrased query.
    - DO NOT USE ANY INFORMATION FROM THE EXAMPLE IN YOUR ANSWER. IT IS MEANT ONLY FOR YOUR UNDERSTANDING.

** Inputs ** :
Original Question: This is natural language question that i want to convert.
Hint: This is hint for the original question for your help.
Required Columns: This is a dictionary in which the keys are table and the values are list of columns. 

** Output **:
The output should be a rephrased version of the question, rewritten to

In [8]:
from src.model.dsx_openai import LLM
from openai import OpenAI
import httpx
from dotenv import load_dotenv
load_dotenv()

# model = 'mixtral-8x7b-instruct-v01'
# model = 'codestral-22b-v0-1'
model = 'llama-3-8b-instruct'

http_client=httpx.Client(verify=False)
client = OpenAI(
    base_url=os.environ['DSX_URL'],
    http_client=http_client,
    api_key=os.environ['DSX_API_KEY']
)
 
llm = LLM(
    client=client,
    model=model,
    gen_params={
        'stream': True,
        'TEMPERATURE': 0,
        'MAX_NEW_TOKENS': 4096
    }
)

  from .autonotebook import tqdm as notebook_tqdm
None of PyTorch, TensorFlow >= 2.0, or Flax have been found. Models won't be available and only tokenizers, configuration and file/data utilities can be used.


In [9]:
completion = llm(
        response_type='single',
        prompts=query_rephrase_prompt
    )

In [10]:
print(completion.result)

Here is the rephrased question:

What is the administrator's email address (refer - schools.AdmEmail1) of the chartered school (refer - frpm.Charter School (Y/N) = 1) with the fewest students enrolled in grades 1 through 12 (refer - frpm.Enrollment (K-12))?
