# RAG From Scratch

![RAG](rag.png)

`1. Packages`

In [None]:
! pip install langchain_community tiktoken langchain-openai langchainhub chromadb langchain

`2. API KEYS`

In [None]:
import os
from dotenv import load_dotenv

# Load environment variables from the .env file
load_dotenv()

LANGCHAIN_API_KEY = os.getenv("LANGCHAIN_API_KEY")
OPENAI_API_KEY = os.getenv("OPENAI_API_KEY")

In [None]:
os.environ['LANGCHAIN_TRACING_V2'] = 'true'
os.environ['LANGCHAIN_ENDPOINT'] = 'https://api.smith.langchain.com'
os.environ['LANGCHAIN_API_KEY'] = LANGCHAIN_API_KEY
os.environ['OPENAI_API_KEY'] = OPENAI_API_KEY

# Part 1: Overview

In [None]:
import bs4 # BeautifulSoup library for web scraping and parsing HTML/XML documents
from langchain import hub # Library for accessing and managing various modules in LangChain
from langchain.text_splitter import RecursiveCharacterTextSplitter # for splitting text into smaller chunks for processing
from langchain_community.document_loaders import WebBaseLoader # Community-maintained library for loading web-based documents
from langchain_community.vectorstores import Chroma # Community-maintained library for creating and managing vector stores
from langchain_core.output_parsers import StrOutputParser # Core library for parsing output in LangChain
from langchain_core.runnables import RunnablePassthrough # Core library for creating pass-through runnables in LangChain
from langchain_openai import ChatOpenAI # Library for using OpenAI's ChatGPT models in LangChain
from langchain_openai import OpenAIEmbeddings # Library for generating and managing OpenAI embeddings in LangChain

### \#### INDEXING \#### 

In [None]:
# Load Documents
loader = WebBaseLoader(
    web_paths=("https://lilianweng.github.io/posts/2023-06-23-agent/",),
    bs_kwargs=dict(
        parse_only=bs4.SoupStrainer(
            class_=("post-content", "post-title", "post-header")
        )
    ),
)
docs = loader.load()
docs

In [None]:
# Split
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
splits = text_splitter.split_documents(docs)
splits[0]

In [None]:
# Embed
vectorstore = Chroma.from_documents(documents=splits,
                                    embedding=OpenAIEmbeddings())
retriever = vectorstore.as_retriever()

#### \#### RETRIEVAL and GENERATION \####

In [None]:
# Prompt
prompt = hub.pull("rlm/rag-prompt")
prompt

In [None]:
# LLM
llm = ChatOpenAI(model_name="gpt-3.5-turbo", temperature=0)
llm

In [None]:
# Post-processing
def format_docs(docs):
    return "\n\n".join(doc.page_content for doc in docs)

In [None]:
# Chain
rag_chain = (
    {"context": retriever | format_docs, "question": RunnablePassthrough()}
    | prompt
    | llm
    | StrOutputParser()
)
rag_chain

In [None]:
# Question
rag_chain.invoke("What is Task Decomposition?")

__Everything we have done above in a single code cell__

In [None]:
import bs4
from langchain import hub
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_community.document_loaders import WebBaseLoader
from langchain_community.vectorstores import Chroma
from langchain_core.output_parsers import StrOutputParser
from langchain_core.runnables import RunnablePassthrough
from langchain_openai import ChatOpenAI, OpenAIEmbeddings

#### INDEXING #### 

# Load Documents
loader = WebBaseLoader(
    web_paths=("https://lilianweng.github.io/posts/2023-06-23-agent/",),
    bs_kwargs=dict(
        parse_only=bs4.SoupStrainer(
            class_=("post-content", "post-title", "post-header")
        )
    ),
)
docs = loader.load()

# Split
text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
splits = text_splitter.split_documents(docs)

# Embed
vectorstore = Chroma.from_documents(documents=splits,
                                    embedding=OpenAIEmbeddings())
retriever = vectorstore.as_retriever()

#### RETRIEVAL and GENERATION ####

# Prompt
prompt = hub.pull("rlm/rag-prompt")

# LLM
llm = ChatOpenAI(model_name="gpt-3.5-turbo", temperature=0)

# Post-processing
def format_docs(docs):
    return "\n\n".join(doc.page_content for doc in docs)

# Chain
rag_chain = (
    {"context": retriever | format_docs, "question": RunnablePassthrough()}
    | prompt
    | llm
    | StrOutputParser()
)

# Question
rag_chain.invoke("What is Task Decomposition?")

# Part 2: Indexing

![title](indexing.png)

In [None]:
# Documents
document = "My favourite pet is a cat."
question = "What kinds of pets do I like?"

How to count tokens using tiktoken ? <br>
__https://github.com/openai/openai-cookbook/blob/main/examples/How_to_count_tokens_with_tiktoken.ipynb__

In [None]:
import tiktoken

def num_tokens_from_string(string: str, encoding_name: str) -> int:
    """Returns the number of tokens in a text string."""
    encoding = tiktoken.get_encoding(encoding_name)
    num_tokens = len(encoding.encode(string))
    return num_tokens

num_tokens_from_string(question, "cl100k_base")

[__Text Embedding Models__](https://python.langchain.com/v0.2/docs/integrations/text_embedding/openai/)

In [None]:
from langchain_openai import OpenAIEmbeddings
embd = OpenAIEmbeddings()
query_result = embd.embed_query(question)
document_result = embd.embed_query(document)
len(query_result), len(document_result)

[Cosine Similarity](https://platform.openai.com/docs/guides/embeddings/use-cases) is recommended for OpenAI embeddings.

__Why we calculated the cosine similarity between query and document ?__ <br>
When you have a query and you want to find the most relevant document or response from a set of documents, you:

1. Embed the Query and Documents: Convert the query and each document into vector representations.
2. Calculate Similarity: Compute the cosine similarity between the query vector and each document vector.
3. Rank Results: Rank the documents based on their similarity scores. The document with the highest similarity score is considered the most relevant to the query.

In [None]:
import numpy as np

def cosine_similarity(vec1, vec2):
    dot_product = np.dot(vec1, vec2)
    norm_vec1 = np.linalg.norm(vec1)
    norm_vec2 = np.linalg.norm(vec2)
    return dot_product / (norm_vec1 * norm_vec2)

similarity = cosine_similarity(query_result, document_result)
print("Cosine Similarity: ", similarity)

In [None]:
#### INDEXING ####

# Load blog
import bs4

from langchain_community.document_loaders import WebBaseLoader
loader = WebBaseLoader(
    web_paths=("https://lilianweng.github.io/posts/2023-06-23-agent/",),
    bs_kwargs=dict(
        parse_only=bs4.SoupStrainer(
            class_=("post-content", "post-title", "post-header")
        )
    ),
)

blog_docs = loader.load()

[__Splitter__](https://python.langchain.com/v0.1/docs/modules/data_connection/document_transformers/recursive_text_splitter/)
> This text splitter is the recommended one for generic text. It is parameterized by a list of characters. It tries to split on them in order until the chunks are small enough. The default list is ["\n\n", "\n", " ", ""]. This has the effect of trying to keep all paragraphs (and then sentences, and then words) together as long as possible, as those would generically seem to be the strongest semantically related pieces of text.

In [None]:
# Splitting
from langchain.text_splitter import RecursiveCharacterTextSplitter

text_splitter = RecursiveCharacterTextSplitter.from_tiktoken_encoder(
    chunk_size=300,
    chunk_overlap=50)

splits = text_splitter.split_documents(blog_docs)

# Part 3: Retrieval

In [None]:
# Index
from langchain_openai import OpenAIEmbeddings
from langchain_community.vectorstores import Chroma

vectorstore = Chroma.from_documents(documents=splits,
                                    embedding=OpenAIEmbeddings())

retriever = vectorstore.as_retriever(search_kwargs={"k": 1}) # returns nearest 1 vector (document)

In [None]:
docs = retriever.invoke("What is Task Decomposition?")

In [None]:
len(docs) # as k=1 we get only 1 doc

In [None]:
docs

# Part 4: Generation

![title](generation.png)

In [None]:
from langchain_openai import ChatOpenAI
from langchain.prompts import ChatPromptTemplate

# prompt
template = """Answer the question based only on the following context:
              {context}
              Question: {question}
           """

prompt = ChatPromptTemplate.from_template(template)
prompt

In [None]:
# llm
llm = ChatOpenAI(model_name="gpt-3.5-turbo", temperature=0)

In [None]:
# chain
chain = prompt | llm

In [None]:
# run 
chain.invoke({"context":docs, "question":"What is Task Decomposition"})

In [None]:
from langchain import hub
prompt_hub_rag = hub.pull("rlm/rag-prompt")
prompt_hub_rag

In [None]:
from langchain_core.output_parsers import StrOutputParser
from langchain_core.runnables import RunnablePassthrough

rag_chain = (
    {"context": retriever, "question": RunnablePassthrough()}
    | prompt
    | llm
    | StrOutputParser()
)

rag_chain.invoke("What is Task Decomposition?")