### Setup

In [1]:
%pip install openai
%pip install langchain
%pip install unstructured
%pip install -U langchain-openai
%pip install langchain langchain-community
%pip install --upgrade --quiet  langchain-pinecone langchain-openai langchain
%pip install --upgrade --quiet pypdf

Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Collecting langchain-openai
  Downloading langchain_openai-0.1.16-py3-none-any.whl.metadata (2.5 kB)
Collecting langchain-core<0.3.0,>=0.2.17 (from langchain-openai)
  Downloading langchain_core-0.2.19-py3-none-any.whl.metadata (6.0 kB)
Downloading langchain_openai-0.1.16-py3-none-any.whl (46 kB)
[2K   [90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━[0m [32m46.1/46.1 kB[0m [31m3.2 MB/s[0m eta [36m0:00:00[0m
[?25hDownloading langchain_core-0.2.19-py3-none-any.whl (366 kB)
[2K   [90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━[0m [32m366.5/366.5 kB[0m [31m16.7 MB/s[0m eta [36m0:00:00[0m
[?25hInstalling collected packages: langchain-core, langchain-openai
  Attempting uninstall: langchain-core
    Found existing installation: langchain-core 0.2.12
    Uninstalling langchain-core-0.2

In [2]:
import langchain
from langchain_openai import OpenAI
from langchain.chains import RetrievalQAWithSourcesChain
from langchain.chains.qa_with_sources.loading import load_qa_with_sources_chain
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_openai import OpenAIEmbeddings
from langchain_pinecone import Pinecone
from langchain_community.document_loaders import PyPDFLoader

### Get the configuration from the configuration file

In [3]:
from dotenv import load_dotenv
import os

load_dotenv()  # take environment variables from .env.

openai_api_key = os.getenv('OPENAI_API_KEY')
pinecone_api_key = os.getenv('PINECONE_API_KEY')

os.environ['OPENAI_API_KEY'] = openai_api_key
os.environ['PINECONE_API_KEY'] = pinecone_api_key

### Extract text from PDF

In [5]:
def extract_text_from_pdf(pdf_path):
    loader = PyPDFLoader(pdf_path)
    documents = loader.load()
    return documents


In [6]:
# Extract text from a PDF file
pdf_path = "./Interim_Budget.pdf"
data = extract_text_from_pdf(pdf_path)
print('Number of documents:', len(data))

Number of documents: 32


### Split text into chunks

In [7]:
def split_data_into_chunks(data, chunk_size=1000, chunk_overlap=200):
    text_splitter = RecursiveCharacterTextSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)
    docs = text_splitter.split_documents(data)
    return docs

In [10]:
# Split the text into chunks
docs = split_data_into_chunks(data)

# Print the number of chunks
print('Number of chunks:', len(docs))

Number of chunks: 57


### Generate embeddings

In [8]:
# Create the embeddings of the chunks using openAIEmbeddings
embeddings = OpenAIEmbeddings()

### Store embeddings in a database

In [11]:
docsearch = Pinecone.from_documents(documents=docs, embedding=embeddings, index_name="qabot")

### Retrive embeddings from the database

In [12]:
query = "Who is finance minister of India?"
docs = docsearch.similarity_search(query)
print('Total number of documents:', len(docs))
print(docs[0].page_content)

Total number of documents: 4
GOVERNMENT OF INDIA
INTERIM BUDGET 2024-2025
SPEECH
OF
NIRMALA SITHARAMAN
MINISTER OF FINANCE
February 1,  2024


### Initialize the language model

In [13]:
llm = OpenAI(temperature=0.7, max_tokens=500)

### Query the System

In [14]:
chain = RetrievalQAWithSourcesChain.from_llm(llm=llm, retriever=docsearch.as_retriever())

In [15]:
query = "Who is finance minister of India?"

langchain.debug=True

chain({"question": query}, return_only_outputs=True)

  warn_deprecated(


[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain] Entering Chain run with input:
[0m{
  "question": "Who is finance minister of India?"
}
[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain > chain:MapReduceDocumentsChain] Entering Chain run with input:
[0m[inputs]
[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain > chain:MapReduceDocumentsChain > chain:LLMChain] Entering Chain run with input:
[0m{
  "input_list": [
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINISTER OF FINANCE\nFebruary 1,  2024",
      "question": "Who is finance minister of India?"
    },
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINISTER OF FINANCE\nFebruary 1,  2024",
      "question": "Who is finance minister of India?"
    },
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINIS

{'answer': ' Nirmala Sitharaman is the finance minister of India.\n',
 'sources': '/Users/satyamvats/Documents/ioco/poc/python/data-science-practice/app/Interim_Budget.pdf'}

### Use conversation buffer to store the conversation history

In [16]:
from langchain.memory import ConversationBufferWindowMemory
from langchain.chains import ConversationChain

# Create a conversation chain with a memory of 5
memory = ConversationBufferWindowMemory(k=5)

convo = ConversationChain(
    llm=llm,
    memory=memory
)

print(convo.prompt.template)

The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully says it does not know.

Current conversation:
{history}
Human: {input}
AI:


  warn_deprecated(


In [17]:
convo.run("Who is finance minister of India?")

  warn_deprecated(


[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "Who is finance minister of India?",
  "history": ""
}
[32;1m[1;3m[llm/start][0m [1m[chain:ConversationChain > llm:OpenAI] Entering LLM run with input:
[0m{
  "prompts": [
    "The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully says it does not know.\n\nCurrent conversation:\n\nHuman: Who is finance minister of India?\nAI:"
  ]
}
[36;1m[1;3m[llm/end][0m [1m[chain:ConversationChain > llm:OpenAI] [1.76s] Exiting LLM run with output:
[0m{
  "generations": [
    [
      {
        "text": " As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance an

' As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance and Corporate Affairs in the Indian government. She has a degree in economics and has worked in the banking sector before entering politics. Is there anything else you would like to know about her or the Indian finance ministry?'

In [18]:
convo.run("What is the budget of India?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "What is the budget of India?",
  "history": "Human: Who is finance minister of India?\nAI:  As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance and Corporate Affairs in the Indian government. She has a degree in economics and has worked in the banking sector before entering politics. Is there anything else you would like to know about her or the Indian finance ministry?"
}
[32;1m[1;3m[llm/start][0m [1m[chain:ConversationChain > llm:OpenAI] Entering LLM run with input:
[0m{
  "prompts": [
    "The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully 

' The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It includes allocations for various sectors such as education, healthcare, infrastructure, and defense. Is there anything else you would like to know about the Indian budget?'

In [20]:
convo.run("When is Interim Budget is passed year 2021?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "When is Interim Budget is passed year 2021?",
  "history": "Human: Who is finance minister of India?\nAI:  As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance and Corporate Affairs in the Indian government. She has a degree in economics and has worked in the banking sector before entering politics. Is there anything else you would like to know about her or the Indian finance ministry?\nHuman: What is the budget of India?\nAI:  The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It includes allocations for various sectors 

' As mentioned earlier, the Interim Budget for the fiscal year 2021-2022 was passed on February 1, 2021. The specific date for the Interim Budget can vary from year to year depending on when the full budget is presented. Is there anything else I can assist you with?'

In [21]:
convo.run("How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?",
  "history": "Human: Who is finance minister of India?\nAI:  As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance and Corporate Affairs in the Indian government. She has a degree in economics and has worked in the banking sector before entering politics. Is there anything else you would like to know about her or the Indian finance ministry?\nHuman: What is the budget of India?\nAI:  The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It incl

" The total amount of Direct Benefit Transfer from the government through PM-Jan Dhan accounts is approximately 1.63 lakh crore Indian rupees. This amount was transferred to over 20 crore beneficiaries during the fiscal year 2020-2021. Is there anything else you would like to know about the Indian government's initiatives for financial inclusion?"

In [40]:
convo.run("How many farmers get direct financial assistance under PM-KISAN SAMMAN Yojana?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How many farmers get direct financial assistance under PM-KISAN SAMMAN Yojana?",
  "history": "Human: Who is finance minister of India?\nAI:  As of 2021, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in 2019 and is the first full-time female finance minister of India. Prior to this, she served as the Minister of Defence and Minister of Commerce and Industry. She holds a degree in economics and has worked in the banking sector before entering politics. \nHuman: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This is an increase from the previous year's budget of 30.42 lakh crore rupees. The budget is presented by the finance minister of India and outlines the government's planned expenditures and revenues for the fiscal year. It is an important document that sets the tone for t

' As of 2021, approximately 9.5 crore farmers have received direct financial assistance under the PM-KISAN SAMMAN Yojana. This scheme provides eligible farmers with an income support of 6,000 rupees per year in three equal installments. The total budget allocated for this scheme in 2021-2022 is 65,000 crore rupees. '

In [22]:
convo.run("How many yuths are trained under 'The Skill India Mission'?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How many yuths are trained under 'The Skill India Mission'?",
  "history": "Human: Who is finance minister of India?\nAI:  As of now, the finance minister of India is Nirmala Sitharaman. She was appointed to this position in May 2019 by Prime Minister Narendra Modi. Prior to this, she served as the Minister of Defence and Minister of State for Finance and Corporate Affairs in the Indian government. She has a degree in economics and has worked in the banking sector before entering politics. Is there anything else you would like to know about her or the Indian finance ministry?\nHuman: What is the budget of India?\nAI:  The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It includes allocations for 

"  As of December 2020, over 1.4 crore youths have been trained under the Skill India Mission. This initiative was launched in 2015 by Prime Minister Narendra Modi with the aim of providing vocational training to the country's youth and increasing their employability. Is there anything else you would like to know about this initiative?"

In [23]:
print(convo.memory.buffer)

Human: What is the budget of India?
AI:  The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It includes allocations for various sectors such as education, healthcare, infrastructure, and defense. Is there anything else you would like to know about the Indian budget?
Human: When is Interim Budget is passed this year?
AI:  The Interim Budget, also known as the Vote on Account, was passed on February 1, 2021. This is a temporary budget that is passed by the government ahead of the full budget to ensure the smooth functioning of the government until the new budget is passed. Is there anything else you would like to know about the Indian budget process?
Human: When is Interim Budget is passed year 2021?
AI:  As mentioned earlier, the Interim Budget for the fiscal year 2021-2022 was passed on February 1, 2021.

In [24]:
convo.run("How Much Loan is sanctioned under PM Mudra Yojana for entrepreneurial aspirations of youths?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How Much Loan is sanctioned under PM Mudra Yojana for entrepreneurial aspirations of youths?",
  "history": "Human: What is the budget of India?\nAI:  The budget of India for the fiscal year 2021-2022 is approximately 34.83 lakh crore Indian rupees, which is equivalent to around 476 billion US dollars. This budget was presented by Finance Minister Nirmala Sitharaman in February 2021. It includes allocations for various sectors such as education, healthcare, infrastructure, and defense. Is there anything else you would like to know about the Indian budget?\nHuman: When is Interim Budget is passed this year?\nAI:  The Interim Budget, also known as the Vote on Account, was passed on February 1, 2021. This is a temporary budget that is passed by the government ahead of the full budget to ensure the smooth functioning of the government until the new budget is passed. Is there anythi

'  Under the PM Mudra Yojana, loans worth over 25 lakh crore Indian rupees have been sanctioned to around 28 crore beneficiaries as of December 2020. This scheme, launched in 2015, aims to provide financial assistance to small and micro-enterprises in the country, including those run by youth entrepreneurs. Is there anything else you would like to know about this scheme?'

In [25]:
print(convo.memory.buffer)

Human: When is Interim Budget is passed this year?
AI:  The Interim Budget, also known as the Vote on Account, was passed on February 1, 2021. This is a temporary budget that is passed by the government ahead of the full budget to ensure the smooth functioning of the government until the new budget is passed. Is there anything else you would like to know about the Indian budget process?
Human: When is Interim Budget is passed year 2021?
AI:  As mentioned earlier, the Interim Budget for the fiscal year 2021-2022 was passed on February 1, 2021. The specific date for the Interim Budget can vary from year to year depending on when the full budget is presented. Is there anything else I can assist you with?
Human: How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?
AI:  The total amount of Direct Benefit Transfer from the government through PM-Jan Dhan accounts is approximately 1.63 lakh crore Indian rupees. This amount was transferred to over 20 crore beneficiar