### Setup

In [25]:
%pip install openai
%pip install langchain
%pip install unstructured
%pip install -U langchain-openai
%pip install langchain langchain-community
%pip install --upgrade --quiet  langchain-pinecone langchain-openai langchain
%pip install --upgrade --quiet pypdf

Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.
Note: you may need to restart the kernel to use updated packages.


In [26]:
import langchain
from langchain_openai import OpenAI
from langchain.chains import RetrievalQAWithSourcesChain
from langchain.chains.qa_with_sources.loading import load_qa_with_sources_chain
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_openai import OpenAIEmbeddings
from langchain_pinecone import Pinecone
from langchain_community.document_loaders import PyPDFLoader

### Get the configuration from the configuration file

In [27]:
from dotenv import load_dotenv
import os

load_dotenv()  # take environment variables from .env.

openai_api_key = os.getenv('OPENAI_API_KEY')
pinecone_api_key = os.getenv('PINECONE_API_KEY')

os.environ['OPENAI_API_KEY'] = openai_api_key
os.environ['PINECONE_API_KEY'] = pinecone_api_key

### Extract text from PDF

In [28]:
def extract_text_from_pdf(pdf_path):
    loader = PyPDFLoader(pdf_path)
    documents = loader.load()
    return documents


In [29]:
# Extract text from a PDF file
pdf_path = "./Interim_Budget.pdf"
data = extract_text_from_pdf(pdf_path)
print('Number of documents:', len(data))

Number of documents: 32


### Split text into chunks

In [30]:
def split_data_into_chunks(data, chunk_size=1000, chunk_overlap=200):
    text_splitter = RecursiveCharacterTextSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)
    docs = text_splitter.split_documents(data)
    return docs

In [31]:
# Split the text into chunks
docs = split_data_into_chunks(data)

# Print the number of chunks
print('Number of chunks:', len(docs))

Number of chunks: 57


### Generate embeddings

In [32]:
# Create the embeddings of the chunks using openAIEmbeddings
embeddings = OpenAIEmbeddings()

### Store embeddings in a database

In [33]:
docsearch = Pinecone.from_documents(documents=docs, embedding=embeddings, index_name="qabot")

### Retrive embeddings from the database

In [34]:
query = "Who is finance minister of India?"
docs = docsearch.similarity_search(query)
print('Total number of documents:', len(docs))
print(docs[0].page_content)

Total number of documents: 4
GOVERNMENT OF INDIA
INTERIM BUDGET 2024-2025
SPEECH
OF
NIRMALA SITHARAMAN
MINISTER OF FINANCE
February 1,  2024


### Initialize the language model

In [35]:
llm = OpenAI(temperature=0.7, max_tokens=500)

### Query the System

In [36]:
chain = RetrievalQAWithSourcesChain.from_llm(llm=llm, retriever=docsearch.as_retriever())

In [37]:
query = "Who is finance minister of India?"

langchain.debug=True

chain({"question": query}, return_only_outputs=True)

[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain] Entering Chain run with input:
[0m{
  "question": "Who is finance minister of India?"
}
[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain > chain:MapReduceDocumentsChain] Entering Chain run with input:
[0m[inputs]
[32;1m[1;3m[chain/start][0m [1m[chain:RetrievalQAWithSourcesChain > chain:MapReduceDocumentsChain > chain:LLMChain] Entering Chain run with input:
[0m{
  "input_list": [
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINISTER OF FINANCE\nFebruary 1,  2024",
      "question": "Who is finance minister of India?"
    },
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINISTER OF FINANCE\nFebruary 1,  2024",
      "question": "Who is finance minister of India?"
    },
    {
      "context": "GOVERNMENT OF INDIA\nINTERIM BUDGET 2024-2025\nSPEECH\nOF\nNIRMALA SITHARAMAN\nMINIS

{'answer': ' Nirmala Sitharaman is the finance minister of India.\n',
 'sources': '/Users/satyamvats/Documents/ioco/poc/python/data-science-practice/app/Interim_Budget.pdf, ./Interim_Budget.pdf'}

### Use conversation buffer to store the conversation history

In [38]:
from langchain.memory import ConversationBufferWindowMemory
from langchain.chains import ConversationChain

# Create a conversation chain with a memory of 5
memory = ConversationBufferWindowMemory(k=5)

convo = ConversationChain(
    llm=llm,
    memory=memory
)

print(convo.prompt.template)

The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully says it does not know.

Current conversation:
{history}
Human: {input}
AI:


In [39]:
convo.run("Who is finance minister of India?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "Who is finance minister of India?",
  "history": ""
}
[32;1m[1;3m[llm/start][0m [1m[chain:ConversationChain > llm:OpenAI] Entering LLM run with input:
[0m{
  "prompts": [
    "The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully says it does not know.\n\nCurrent conversation:\n\nHuman: Who is finance minister of India?\nAI:"
  ]
}
[36;1m[1;3m[llm/end][0m [1m[chain:ConversationChain > llm:OpenAI] [2.05s] Exiting LLM run with output:
[0m{
  "generations": [
    [
      {
        "text": " The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian governmen

' The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?'

In [40]:
convo.run("What is the budget of India?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "What is the budget of India?",
  "history": "Human: Who is finance minister of India?\nAI:  The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?"
}
[32;1m[1;3m[llm/start][0m [1m[chain:ConversationChain > llm:OpenAI] Entering LLM run with input:
[0m{
  "prompts": [
    "The following is a friendly conversation between a human and an AI. The AI is talkative and provides lots of specific details from its context. If the AI does not know the answer to a question, it truthfully says it does not know.\n\nCurr

' As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for presenting and managing the budget, which is approved by the Indian Parliament. Is there anything else you would like to know about the budget of India?'

In [41]:
convo.run("When is Interim Budget is passed year 2021?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "When is Interim Budget is passed year 2021?",
  "history": "Human: Who is finance minister of India?\nAI:  The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?\nHuman: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for presenting and managing the budget, wh

' The Interim Budget for 2021 was passed on February 1, 2021. This is a temporary budget that is passed before the general elections, to keep the government functioning until a new budget is presented by the new government. It is also known as a vote on account and only covers essential expenses for a short period of time. Is there anything else you would like to know about the Interim Budget?'

In [42]:
convo.run("How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?",
  "history": "Human: Who is finance minister of India?\nAI:  The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?\nHuman: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for p

' As of March 2021, the total amount transferred through Direct Benefit Transfer from the Government using PM-Jan Dhan accounts is approximately 3.79 lakh crore rupees. This scheme was launched in 2014 by the Indian government to transfer subsidies and other benefits directly to the beneficiaries through their Jan Dhan accounts. It has been successful in reducing corruption and ensuring efficient delivery of government aid. Is there anything else you would like to know about this scheme or the amount transferred through it?'

In [43]:
convo.run("How many farmers get direct financial assistance under PM-KISAN SAMMAN Yojana?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How many farmers get direct financial assistance under PM-KISAN SAMMAN Yojana?",
  "history": "Human: Who is finance minister of India?\nAI:  The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?\nHuman: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for pre

' As of March 2021, approximately 11.43 crore farmers have received direct financial assistance under the PM-KISAN SAMMAN Yojana. This scheme was launched in 2018 to provide income support to small and marginal farmers by transferring 6000 rupees per year directly to their bank accounts. The amount is given in three installments and has been beneficial for farmers across the country. Is there anything else you would like to know about this scheme or the number of farmers who have received assistance through it?'

In [44]:
convo.run("How many yuths are trained under 'The Skill India Mission'?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How many yuths are trained under 'The Skill India Mission'?",
  "history": "Human: Who is finance minister of India?\nAI:  The current finance minister of India is Nirmala Sitharaman. She assumed office on May 31, 2019. Prior to this, she served as the Minister of Defence and the Minister of State for Finance and Corporate Affairs in the Indian government. She is a member of the Bharatiya Janata Party and has a background in economics and management. Is there anything else you would like to know about her or the role of finance minister in India?\nHuman: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for presenting and managin

"  As of March 2021, approximately 1.13 crore youth have been trained under 'The Skill India Mission'. This initiative was launched in 2015 to provide skill training to the youth of India and enhance their employability. The program offers training in various sectors such as construction, retail, healthcare, and hospitality, among others. It has been successful in improving the skillset of the Indian workforce and promoting job opportunities. Is there anything else you would like to know about this mission or the number of youth trained under it?"

In [45]:
print(convo.memory.buffer)

Human: What is the budget of India?
AI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for presenting and managing the budget, which is approved by the Indian Parliament. Is there anything else you would like to know about the budget of India?
Human: When is Interim Budget is passed year 2021?
AI:  The Interim Budget for 2021 was passed on February 1, 2021. This is a temporary budget that is passed before the general elections, to keep the government functioning until a new budget is presented by the new government. It is also known as a vote on account and only covers essential expenses for a short period of time. Is there anything else you would like to know about the Interim Budget?
Human: How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accou

In [46]:
convo.run("How Much Loan is sanctioned under PM Mudra Yojana for entrepreneurial aspirations of youths?")

[32;1m[1;3m[chain/start][0m [1m[chain:ConversationChain] Entering Chain run with input:
[0m{
  "input": "How Much Loan is sanctioned under PM Mudra Yojana for entrepreneurial aspirations of youths?",
  "history": "Human: What is the budget of India?\nAI:  As of 2021, the budget of India is approximately 34.83 lakh crore rupees. This includes both the central and state budgets. The budget is allocated towards various sectors such as defense, education, healthcare, and infrastructure. The finance minister is responsible for presenting and managing the budget, which is approved by the Indian Parliament. Is there anything else you would like to know about the budget of India?\nHuman: When is Interim Budget is passed year 2021?\nAI:  The Interim Budget for 2021 was passed on February 1, 2021. This is a temporary budget that is passed before the general elections, to keep the government functioning until a new budget is presented by the new government. It is also known as a vote on acco

' As of March 2021, approximately 28.68 lakh crore rupees have been sanctioned under PM Mudra Yojana for the entrepreneurial aspirations of youth. This scheme was launched in 2015 to provide financial assistance to small and micro enterprises. It offers loans up to 10 lakh rupees to individuals, including youth, to start or expand their businesses. It has been instrumental in promoting entrepreneurship and creating job opportunities in India. Is there anything else you would like to know about this scheme or the amount sanctioned under it?'

In [47]:
print(convo.memory.buffer)

Human: When is Interim Budget is passed year 2021?
AI:  The Interim Budget for 2021 was passed on February 1, 2021. This is a temporary budget that is passed before the general elections, to keep the government functioning until a new budget is presented by the new government. It is also known as a vote on account and only covers essential expenses for a short period of time. Is there anything else you would like to know about the Interim Budget?
Human: How mush Direct Benefit Transfer from the Government using PM-Jan Dhan accounts?
AI:  As of March 2021, the total amount transferred through Direct Benefit Transfer from the Government using PM-Jan Dhan accounts is approximately 3.79 lakh crore rupees. This scheme was launched in 2014 by the Indian government to transfer subsidies and other benefits directly to the beneficiaries through their Jan Dhan accounts. It has been successful in reducing corruption and ensuring efficient delivery of government aid. Is there anything else you wou