In [58]:
!pip install langchain



In [59]:
from langchain.llms import GooglePalm

In [60]:
!pip install -U -q google.generativeai
!pip install -q chromadb

In [61]:
# creating an object for our LLM

api_key = 'your_api_key'

llm = GooglePalm(google_api_key = api_key, temperature = 0.2)

poem = llm("write a poem on my love for ramen")
print(poem)

**Ode to Ramen**

Ramen, oh ramen,
You are my favorite food.
Your broth is so flavorful,
And your noodles are so chewy.

I love the way you make me feel warm and cozy,
Like I'm wrapped in a blanket on a cold winter day.
You're the perfect comfort food,
And I can't get enough of you.

I love the way you smell,
So inviting and delicious.
And I love the way you taste,
So umami and satisfying.

Ramen, oh ramen,
You are my one true love.
I will always cherish you,
And I will never stop eating you.

You are the best food in the world,
And I will never forget you.

Thank you for being my favorite food,
Ramen.


In [62]:
!pip install pymysql



In [63]:
!pip install mysql-connector-python



In [64]:
!pip install langchain_experimental



In [65]:
from langchain.utilities import SQLDatabase
from langchain_experimental.sql import SQLDatabaseChain

In [66]:
import pymysql
import mysql.connector

import sqlalchemy as sal

from sqlalchemy import create_engine, MetaData, Table

import pyodbc

import os
import pandas as pd

In [67]:
db_host = os.getenv('MYSQL_HOST', 'localhost')
db_port = os.getenv('MYSQL_PORT', 3306)
db_user = os.getenv('MYSQL_USER', 'root')
db_password = os.getenv('MYSQL_PASSWORD', 'root')
db_name = os.getenv('MYSQL_DATABASE', 'atliq_tshirts')

In [68]:
db = SQLDatabase.from_uri(f"mysql+pymysql://{db_user}:{db_password}@{db_host}/{db_name}",sample_rows_in_table_info=3)

In [69]:
print(db.table_info)


CREATE TABLE discounts (
	discount_id INTEGER NOT NULL AUTO_INCREMENT, 
	t_shirt_id INTEGER NOT NULL, 
	pct_discount DECIMAL(5, 2), 
	PRIMARY KEY (discount_id), 
	CONSTRAINT discounts_ibfk_1 FOREIGN KEY(t_shirt_id) REFERENCES t_shirts (t_shirt_id), 
	CONSTRAINT discounts_chk_1 CHECK ((`pct_discount` between 0 and 100))
)COLLATE utf8mb4_0900_ai_ci ENGINE=InnoDB DEFAULT CHARSET=utf8mb4

/*
3 rows from discounts table:
discount_id	t_shirt_id	pct_discount
1	1	10.00
2	2	15.00
3	3	20.00
*/


CREATE TABLE t_shirts (
	t_shirt_id INTEGER NOT NULL AUTO_INCREMENT, 
	brand ENUM('Van Huesen','Levi','Nike','Adidas') NOT NULL, 
	color ENUM('Red','Blue','Black','White') NOT NULL, 
	size ENUM('XS','S','M','L','XL') NOT NULL, 
	price INTEGER, 
	stock_quantity INTEGER NOT NULL, 
	PRIMARY KEY (t_shirt_id), 
	CONSTRAINT t_shirts_chk_1 CHECK ((`price` between 10 and 50))
)COLLATE utf8mb4_0900_ai_ci ENGINE=InnoDB DEFAULT CHARSET=utf8mb4

/*
3 rows from t_shirts table:
t_shirt_id	brand	color	size	price	stock

# Creating an SQLDatabaseChain

In [71]:
db_chain = SQLDatabaseChain.from_llm(llm, db, verbose = True)
question1 = db_chain.run("How many t-shirts do we have left for nike in extra small size and white color?")



[1m> Entering new SQLDatabaseChain chain...[0m
How many t-shirts do we have left for nike in extra small size and white color?
SQLQuery:[32;1m[1;3mSELECT stock_quantity FROM t_shirts WHERE brand = 'Nike' AND color = 'White' AND size = 'XS'[0m
SQLResult: [33;1m[1;3m[(79,)][0m
Answer:[32;1m[1;3m79[0m
[1m> Finished chain.[0m


In [72]:
question1

'79'

In [77]:
# a complex query
question2 = db_chain.run("How much is the price of the inventory for all small size t-shirts?")



[1m> Entering new SQLDatabaseChain chain...[0m
How much is the price of the inventory for all small size t-shirts?
SQLQuery:[32;1m[1;3mSELECT SUM(price) FROM t_shirts WHERE size = 'S'[0m
SQLResult: [33;1m[1;3m[(Decimal('360'),)][0m
Answer:[32;1m[1;3m360[0m
[1m> Finished chain.[0m


In [78]:
# storing the correct query
question2 = db_chain.run("SELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S'")



[1m> Entering new SQLDatabaseChain chain...[0m
SELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S'
SQLQuery:[32;1m[1;3mSELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S'[0m
SQLResult: [33;1m[1;3m[(Decimal('19358'),)][0m
Answer:[32;1m[1;3m19358[0m
[1m> Finished chain.[0m


In [80]:
# a more complex query
# throws error
question3 = db_chain.run("If we have to sell all the Levi’s T-shirts today with discounts applied. How much revenue our store will generate (post discounts)?")



[1m> Entering new SQLDatabaseChain chain...[0m
If we have to sell all the Levi’s T-shirts today with discounts applied. How much revenue our store will generate (post discounts)?
SQLQuery:[32;1m[1;3mSELECT SUM(price * (1 - pct_discount)) FROM t_shirts JOIN discounts ON t_shirts.t_shirt_id = discounts.t_shirt_id WHERE brand = 'Levi' AND CURDATE() BETWEEN discounts.start_date AND discounts.end_date[0m

OperationalError: (pymysql.err.OperationalError) (1054, "Unknown column 'discounts.start_date' in 'where clause'")
[SQL: SELECT SUM(price * (1 - pct_discount)) FROM t_shirts JOIN discounts ON t_shirts.t_shirt_id = discounts.t_shirt_id WHERE brand = 'Levi' AND CURDATE() BETWEEN discounts.start_date AND discounts.end_date]
(Background on this error at: https://sqlalche.me/e/14/e3q8)

In [81]:
# storing the correct query

sql_code = """
select sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Levi'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id
 """

question3 = db_chain.run(sql_code)



[1m> Entering new SQLDatabaseChain chain...[0m

select sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Levi'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id
 
SQLQuery:[32;1m[1;3mselect sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Levi'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id[0m
SQLResult: [33;1m[1;3m[(Decimal('23996.000000'),)][0m
Answer:[32;1m[1;3m23996.0[0m
[1m> Finished chain.[0m


In [93]:
# wrong
question4 = db_chain.run("If we have to sell all the Levi’s T-shirts today. How much revenue our store will generate without discount?")



[1m> Entering new SQLDatabaseChain chain...[0m
If we have to sell all the Levi’s T-shirts today. How much revenue our store will generate without discount?
SQLQuery:[32;1m[1;3mSELECT SUM(price) FROM t_shirts WHERE brand = 'Levi'[0m
SQLResult: [33;1m[1;3m[(Decimal('533'),)][0m
Answer:[32;1m[1;3m533[0m
[1m> Finished chain.[0m


In [94]:
question4 = db_chain.run("SELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Levi'")



[1m> Entering new SQLDatabaseChain chain...[0m
SELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Levi'
SQLQuery:[32;1m[1;3mSELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Levi'[0m
SQLResult: [33;1m[1;3m[(Decimal('23996'),)][0m
Answer:[32;1m[1;3m23996[0m
[1m> Finished chain.[0m


In [91]:
# wrong
question5 = db_chain.run("How many white color Levi's t shirts are available?")



[1m> Entering new SQLDatabaseChain chain...[0m
How many white color Levi's t shirts are available?
SQLQuery:[32;1m[1;3mSELECT stock_quantity FROM t_shirts WHERE brand = 'Levi' AND color = 'White'[0m
SQLResult: [33;1m[1;3m[(28,), (48,)][0m
Answer:[32;1m[1;3m48[0m
[1m> Finished chain.[0m


In [92]:
question5 = db_chain.run("SELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'")



[1m> Entering new SQLDatabaseChain chain...[0m
SELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'
SQLQuery:[32;1m[1;3mSELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'[0m
SQLResult: [33;1m[1;3m[(Decimal('76'),)][0m
Answer:[32;1m[1;3m76[0m
[1m> Finished chain.[0m


# FEW SHOT LEARNING

In [97]:
few_shots = [
    {
        'Question' : "How many t-shirts do we have left for nike in extra small size and white color?",
        'SQLQuery' : "SELECT stock_quantity FROM t_shirts WHERE brand = 'Nike' AND color = 'White' AND size = 'XS'",
        'SQLResult' : "Result of the SQL query",
        'Answer' : question1
    },
    {
        'Question' : "How much is the price of the inventory for all small size t-shirts?",
        'SQLQuery' : "SELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S'",
        'SQLResult' : "Result of the SQL query",
        'Answer' : question2
    },
    {
        'Question' : "If we have to sell all the Levi’s T-shirts today with discounts applied. How much revenue our store will generate (post discounts)?",
        'SQLQuery' : """SELECT sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Levi'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id
 """,
        'SQLResult' : "Result of the SQL query",
        'Answer' : question3
    },
    {
        'Question' : "If we have to sell all the Levi’s T-shirts today. How much revenue our store will generate without discount?",
        'SQLQuery' : "SELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Levi'",
        'SQLResult' : "Result of the SQL query",
        'Answer' : question4
    },
    {
        'Question' : "How many white color Levi's t shirts are available?",
        'SQLQuery' : "SELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'",
        'SQLResult' : "Result of the SQL query",
        'Answer' : question5
    }
]

In [99]:
!pip install sentence-transformers

Collecting sentence-transformers
  Downloading sentence-transformers-2.2.2.tar.gz (85 kB)
[2K     [90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━[0m [32m86.0/86.0 kB[0m [31m1.6 MB/s[0m eta [36m0:00:00[0ma [36m0:00:01[0m
[?25h  Preparing metadata (setup.py) ... [?25ldone
Collecting torch>=1.6.0 (from sentence-transformers)
  Obtaining dependency information for torch>=1.6.0 from https://files.pythonhosted.org/packages/1e/86/477ec85bf1f122931f00a2f3889ed9322c091497415a563291ffc119dacc/torch-2.1.2-cp311-none-macosx_11_0_arm64.whl.metadata
  Downloading torch-2.1.2-cp311-none-macosx_11_0_arm64.whl.metadata (25 kB)
Collecting torchvision (from sentence-transformers)
  Obtaining dependency information for torchvision from https://files.pythonhosted.org/packages/ef/a2/f16cac894c4c71585b3411707502ed8d607945fb4a695857621565bd728d/torchvision-0.16.2-cp311-cp311-macosx_11_0_arm64.whl.metadata
  Downloading torchvision-0.16.2-cp311-cp311-macosx_11_0_arm64.whl.metadata (6.6 kB)
Collecting

In [100]:
# generating embeddings

from langchain.embeddings import HuggingFaceEmbeddings

embeddings = HuggingFaceEmbeddings(model_name = 'sentence-transformers/all-MiniLM-L6-v2')

Downloading .gitattributes:   0%|          | 0.00/1.18k [00:00<?, ?B/s]

Downloading 1_Pooling/config.json:   0%|          | 0.00/190 [00:00<?, ?B/s]

Downloading README.md:   0%|          | 0.00/10.6k [00:00<?, ?B/s]

Downloading config.json:   0%|          | 0.00/612 [00:00<?, ?B/s]

Downloading (…)ce_transformers.json:   0%|          | 0.00/116 [00:00<?, ?B/s]

Downloading data_config.json:   0%|          | 0.00/39.3k [00:00<?, ?B/s]

Downloading pytorch_model.bin:   0%|          | 0.00/90.9M [00:00<?, ?B/s]

Downloading (…)nce_bert_config.json:   0%|          | 0.00/53.0 [00:00<?, ?B/s]

Downloading (…)cial_tokens_map.json:   0%|          | 0.00/112 [00:00<?, ?B/s]

Downloading tokenizer.json:   0%|          | 0.00/466k [00:00<?, ?B/s]

Downloading tokenizer_config.json:   0%|          | 0.00/350 [00:00<?, ?B/s]

Downloading train_script.py:   0%|          | 0.00/13.2k [00:00<?, ?B/s]

Downloading vocab.txt:   0%|          | 0.00/232k [00:00<?, ?B/s]

Downloading modules.json:   0%|          | 0.00/349 [00:00<?, ?B/s]

In [102]:
# merging few_shots as a string

to_vectorize = [" ".join(example.values()) for example in few_shots]
to_vectorize

["How many t-shirts do we have left for nike in extra small size and white color? SELECT stock_quantity FROM t_shirts WHERE brand = 'Nike' AND color = 'White' AND size = 'XS' Result of the SQL query 79",
 "How much is the price of the inventory for all small size t-shirts? SELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S' Result of the SQL query 19358",
 "If we have to sell all the Levi’s T-shirts today with discounts applied. How much revenue our store will generate (post discounts)? SELECT sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from\n(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Levi'\ngroup by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id\n  Result of the SQL query 23996.0",
 "If we have to sell all the Levi’s T-shirts today. How much revenue our store will generate without discount? SELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Levi' R

In [103]:
# creating the vector database

from langchain.vectorstores import Chroma

vectorstore = Chroma.from_texts(to_vectorize, embedding = embeddings, metadatas = few_shots)

In [108]:
# example selector

from langchain.prompts import SemanticSimilarityExampleSelector

example_selector = SemanticSimilarityExampleSelector(vectorstore = vectorstore, k = 2)

example_selector.select_examples({"Question": "How many Adidas T shirts I have left in my store?"})

[{'Answer': '79',
  'Question': 'How many t-shirts do we have left for nike in extra small size and white color?',
  'SQLQuery': "SELECT stock_quantity FROM t_shirts WHERE brand = 'Nike' AND color = 'White' AND size = 'XS'",
  'SQLResult': 'Result of the SQL query'},
 {'Answer': '76',
  'Question': "How many white color Levi's t shirts are available?",
  'SQLQuery': "SELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'",
  'SQLResult': 'Result of the SQL query'}]

# Instructor MySQL_Prompt from Lanchain

In [110]:
# custom mysql prompt that instructs the llm

from langchain.chains.sql_database.prompt import PROMPT_SUFFIX, _mysql_prompt

print(_mysql_prompt)
print(PROMPT_SUFFIX)

You are a MySQL expert. Given an input question, first create a syntactically correct MySQL query to run, then look at the results of the query and return the answer to the input question.
Unless the user specifies in the question a specific number of examples to obtain, query for at most {top_k} results using the LIMIT clause as per MySQL. You can order the results to return the most informative data in the database.
Never query for all columns from a table. You must query only the columns that are needed to answer the question. Wrap each column name in backticks (`) to denote them as delimited identifiers.
Pay attention to use only the column names you can see in the tables below. Be careful to not query for columns that do not exist. Also, pay attention to which column is in which table.
Pay attention to use CURDATE() function to get the current date, if the question involves "today".

Use the following format:

Question: Question here
SQLQuery: SQL Query to run
SQLResult: Result of

# Creating Prompt Templates

In [111]:
# normal prompt template

from langchain.prompts.prompt import PromptTemplate

example_prompt = PromptTemplate(
    input_variables=["Question", "SQLQuery", "SQLResult","Answer",],
    template="\nQuestion: {Question}\nSQLQuery: {SQLQuery}\nSQLResult: {SQLResult}\nAnswer: {Answer}",
)

In [113]:
# few-shot prompt template

from langchain.prompts import FewShotPromptTemplate

few_shot_prompt = FewShotPromptTemplate(
    example_selector=example_selector,
    example_prompt=example_prompt,
    prefix=_mysql_prompt,
    suffix=PROMPT_SUFFIX,
    input_variables=["input", "table_info", "top_k"], #These variables are used in the prefix and suffix
)

# Creating a new SQL DB Chain that accounts for the few-shots as well

In [114]:
new_chain = SQLDatabaseChain.from_llm(llm, db, verbose = True, prompt = few_shot_prompt)

In [116]:
# correct question5
new_chain("How many white color Levi's t shirts are available?")



[1m> Entering new SQLDatabaseChain chain...[0m
How many white color Levi's t shirts are available?
SQLQuery:[32;1m[1;3mSELECT SUM(stock_quantity) FROM t_shirts WHERE brand='Levi' AND color='White'[0m
SQLResult: [33;1m[1;3m[(Decimal('76'),)][0m
Answer:[32;1m[1;3m76[0m
[1m> Finished chain.[0m


{'query': "How many white color Levi's t shirts are available?",
 'result': '76'}

In [118]:
# correct question2
new_chain("How much is the price of the inventory for all small size t-shirts?")



[1m> Entering new SQLDatabaseChain chain...[0m
How much is the price of the inventory for all small size t-shirts?
SQLQuery:[32;1m[1;3mSELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='S'[0m
SQLResult: [33;1m[1;3m[(Decimal('19358'),)][0m
Answer:[32;1m[1;3m19358[0m
[1m> Finished chain.[0m


{'query': 'How much is the price of the inventory for all small size t-shirts?',
 'result': '19358'}

In [120]:
# new query
# correct
new_chain("How much is the price of all the extra small size t-shirts?")



[1m> Entering new SQLDatabaseChain chain...[0m
How much is the price of all the extra small size t-shirts?
SQLQuery:[32;1m[1;3mSELECT SUM(price*stock_quantity) FROM t_shirts WHERE size='XS'[0m
SQLResult: [33;1m[1;3m[(Decimal('19843'),)][0m
Answer:[32;1m[1;3m19843[0m
[1m> Finished chain.[0m


{'query': 'How much is the price of all the extra small size t-shirts?',
 'result': '19843'}

In [121]:
# a complex query
new_chain("If we have to sell all the Nike’s T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?")



[1m> Entering new SQLDatabaseChain chain...[0m
If we have to sell all the Nike’s T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?
SQLQuery:[32;1m[1;3mSELECT sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Nike'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id[0m
SQLResult: [33;1m[1;3m[(Decimal('21744.250000'),)][0m
Answer:[32;1m[1;3m21744.25[0m
[1m> Finished chain.[0m


{'query': 'If we have to sell all the Nike’s T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?',
 'result': '21744.25'}

In [122]:
# complex query
new_chain("If we have to sell all the Van Heuson T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?")



[1m> Entering new SQLDatabaseChain chain...[0m
If we have to sell all the Van Heuson T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?
SQLQuery:[32;1m[1;3mSELECT sum(a.total_amount * ((100-COALESCE(discounts.pct_discount,0))/100)) as total_revenue from
(select sum(price*stock_quantity) as total_amount, t_shirt_id from t_shirts where brand = 'Van Huesen'
group by t_shirt_id) a left join discounts on a.t_shirt_id = discounts.t_shirt_id[0m
SQLResult: [33;1m[1;3m[(Decimal('25154.600000'),)][0m
Answer:[32;1m[1;3m25154.6[0m
[1m> Finished chain.[0m


{'query': 'If we have to sell all the Van Heuson T-shirts today with discounts applied. How much revenue  our store will generate (post discounts)?',
 'result': '25154.6'}

In [123]:
# complex query
new_chain.run('How much revenue  our store will generate by selling all Van Heuson TShirts without discount?')



[1m> Entering new SQLDatabaseChain chain...[0m
How much revenue  our store will generate by selling all Van Heuson TShirts without discount?
SQLQuery:[32;1m[1;3mSELECT SUM(price * stock_quantity) FROM t_shirts WHERE brand = 'Van Huesen'[0m
SQLResult: [33;1m[1;3m[(Decimal('25934'),)][0m
Answer:[32;1m[1;3m25934[0m
[1m> Finished chain.[0m


'25934'