### RAG pipeline with Multiple Data Sources

In [32]:
from langchain_community.tools import WikipediaQueryRun
from langchain_community.utilities import WikipediaAPIWrapper

# Wikipedia tool
api_wrapper = WikipediaAPIWrapper(top_k_results=1, doc_content_chars_max=200)
wiki_tool = WikipediaQueryRun(api_wrapper=api_wrapper)
wiki_tool.name

'wikipedia'

In [33]:
from langchain_community.document_loaders import WebBaseLoader
from langchain_community.vectorstores import FAISS
from langchain_openai import OpenAIEmbeddings
from langchain_text_splitters import RecursiveCharacterTextSplitter

# Loading weburl document using WebBaseLoader
loader = WebBaseLoader("https://docs.smith.langchain.com/")
docs = loader.load()

In [34]:
docs

[Document(metadata={'source': 'https://docs.smith.langchain.com/', 'title': 'Get started with LangSmith | 🦜️🛠️ LangSmith', 'description': 'LangSmith is a platform for building production-grade LLM applications.', 'language': 'en'}, page_content="\n\n\n\n\nGet started with LangSmith | 🦜️🛠️ LangSmith\n\n\n\n\n\n\n\n\nSkip to main contentOur Building Ambient Agents with LangGraph course is now available on LangChain Academy!API ReferenceRESTPythonJS/TSSearchRegionUSEUGo to AppGet StartedObservabilityEvaluationPrompt EngineeringDeployment (LangGraph Platform)AdministrationSelf-hostingPricingReferenceCloud architecture and scalabilityAuthz and AuthnAuthentication methodsdata_formatsEvaluationDataset transformationsRegions FAQsdk_referenceGet StartedOn this pageGet started with LangSmith\nLangSmith is a platform for building production-grade LLM applications.\nIt allows you to closely monitor and evaluate your application, so you can ship quickly and with confidence.\nObservabilityAnalyze tr

In [35]:
# splitting loaded documents into chunk sizes
documents = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200).split_documents(docs)
vectordb = FAISS.from_documents(documents, OpenAIEmbeddings())
retriever = vectordb.as_retriever()
retriever

VectorStoreRetriever(tags=['FAISS', 'OpenAIEmbeddings'], vectorstore=<langchain_community.vectorstores.faiss.FAISS object at 0x0000027CD971B1A0>, search_kwargs={})

In [36]:
#  Creating retriever tool 
from langchain.tools.retriever import create_retriever_tool

retriever_tool = create_retriever_tool(retriever=retriever, 
                                       name="langsmith_search", 
                                       description="Search for information about LangSmith. For any question about LangSmith")
retriever_tool.name

'langsmith_search'

In [37]:
# Arxiv Tool
from langchain_community.utilities import ArxivAPIWrapper
from langchain_community.tools import ArxivQueryRun

arxiv_wrapper = ArxivAPIWrapper(top_k_results=1, doc_content_chars_max=200)
arxiv_tool = ArxivQueryRun(api_wrapper=arxiv_wrapper)
arxiv_tool.name

'arxiv'

In [38]:
tools = [wiki_tool, arxiv_tool, retriever_tool]
tools

[WikipediaQueryRun(api_wrapper=WikipediaAPIWrapper(wiki_client=<module 'wikipedia' from 'c:\\Users\\Sur Ox\\Documents\\GitHub\\Langchain\\chatbot_env\\Lib\\site-packages\\wikipedia\\__init__.py'>, top_k_results=1, lang='en', load_all_available_meta=False, doc_content_chars_max=200)),
 ArxivQueryRun(api_wrapper=ArxivAPIWrapper(arxiv_search=<class 'arxiv.Search'>, arxiv_exceptions=(<class 'arxiv.ArxivError'>, <class 'arxiv.UnexpectedEmptyPageError'>, <class 'arxiv.HTTPError'>), top_k_results=1, ARXIV_MAX_QUERY_LENGTH=300, continue_on_failure=False, load_max_docs=100, load_all_available_meta=False, doc_content_chars_max=200)),
 Tool(name='langsmith_search', description='Search for information about LangSmith. For any question about LangSmith', args_schema=<class 'langchain_core.tools.retriever.RetrieverInput'>, func=functools.partial(<function _get_relevant_documents at 0x0000027CBCE9AD40>, retriever=VectorStoreRetriever(tags=['FAISS', 'OpenAIEmbeddings'], vectorstore=<langchain_community

In [39]:
from dotenv import load_dotenv
from langchain_openai import ChatOpenAI
from langchain_community.llms import Ollama
import os

load_dotenv()




# # Utilising Ollama as my llm model
# # Load Ollama llama2 version model
# llm = Ollama(model="llama2")
# llm


# Load oenai api key
os.environ['OPENAI_API_KEY'] = os.getenv("OPENAI_API_KEY")
llm = ChatOpenAI(model="gpt-3.5-turbo-0125", temperature=0)
llm


ChatOpenAI(client=<openai.resources.chat.completions.completions.Completions object at 0x0000027CDAC67680>, async_client=<openai.resources.chat.completions.completions.AsyncCompletions object at 0x0000027CDACA2300>, root_client=<openai.OpenAI object at 0x0000027CDAC678F0>, root_async_client=<openai.AsyncOpenAI object at 0x0000027CDAC669C0>, model_name='gpt-3.5-turbo-0125', temperature=0.0, model_kwargs={}, openai_api_key=SecretStr('**********'))

In [40]:
from langchain import hub
# from langchain.llms.fireworks import Fireworks


# # Pull down the prompt (llama)
# prompt = hub.pull("yeyuan/baseline-qa-llama2")
# prompt.messages

# Get the prompt to use 
prompt = hub.pull("hwchase17/openai-functions-agent")
prompt.messages




[SystemMessagePromptTemplate(prompt=PromptTemplate(input_variables=[], input_types={}, partial_variables={}, template='You are a helpful assistant'), additional_kwargs={}),
 MessagesPlaceholder(variable_name='chat_history', optional=True),
 HumanMessagePromptTemplate(prompt=PromptTemplate(input_variables=['input'], input_types={}, partial_variables={}, template='{input}'), additional_kwargs={}),
 MessagesPlaceholder(variable_name='agent_scratchpad')]

In [41]:
# AGENTS
"""
LangChain Agent's function is to serve as an intelligent, autonomous controller that uses 
an LLM to reason, plan, and execute a series of actions with the help of various tools to 
fulfill a complex request.
"""

from langchain.agents import create_openai_tools_agent
agent = create_openai_tools_agent(llm, tools, prompt)

In [42]:
# Agent Executer
from langchain.agents import AgentExecutor
agent_executer = AgentExecutor(agent=agent, tools=tools, verbose=True)
agent_executer

AgentExecutor(verbose=True, agent=RunnableMultiActionAgent(runnable=RunnableAssign(mapper={
  agent_scratchpad: RunnableLambda(lambda x: format_to_openai_tool_messages(x['intermediate_steps']))
})
| ChatPromptTemplate(input_variables=['agent_scratchpad', 'input'], optional_variables=['chat_history'], input_types={'chat_history': list[typing.Annotated[typing.Union[typing.Annotated[langchain_core.messages.ai.AIMessage, Tag(tag='ai')], typing.Annotated[langchain_core.messages.human.HumanMessage, Tag(tag='human')], typing.Annotated[langchain_core.messages.chat.ChatMessage, Tag(tag='chat')], typing.Annotated[langchain_core.messages.system.SystemMessage, Tag(tag='system')], typing.Annotated[langchain_core.messages.function.FunctionMessage, Tag(tag='function')], typing.Annotated[langchain_core.messages.tool.ToolMessage, Tag(tag='tool')], typing.Annotated[langchain_core.messages.ai.AIMessageChunk, Tag(tag='AIMessageChunk')], typing.Annotated[langchain_core.messages.human.HumanMessageChunk, Tag

In [44]:
query = agent_executer.invoke({"input": "Tell me about Langsmith"})
query.get('output') 



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `wikipedia` with `{'query': 'LangSmith'}`


[0m[36;1m[1;3mPage: LangChain
Summary: LangChain is a software framework that helps facilitate the integration of large language models (LLMs) into applications. As a language model integration framework, LangChain[0m[32;1m[1;3mLangSmith is a software framework called LangChain that helps integrate large language models (LLMs) into applications. It facilitates the integration of language models into various applications.[0m

[1m> Finished chain.[0m


'LangSmith is a software framework called LangChain that helps integrate large language models (LLMs) into applications. It facilitates the integration of language models into various applications.'

In [None]:
query = agent_executer.invoke({"input": "Tell me about machine learning"})
query.get('output')



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `wikipedia` with `{'query': 'Machine Learning'}`


[0m[36;1m[1;3mPage: Machine learning
Summary: Machine learning (ML) is a field of study in artificial intelligence concerned with the development and study of statistical algorithms that can learn from data and gen[0m[32;1m[1;3mMachine learning (ML) is a field of study in artificial intelligence concerned with the development and study of statistical algorithms that can learn from data and generate insights or predictions. It involves creating models that can automatically learn and improve from experience without being explicitly programmed. Machine learning is used in various applications such as image recognition, natural language processing, and recommendation systems.[0m

[1m> Finished chain.[0m


'Machine learning (ML) is a field of study in artificial intelligence concerned with the development and study of statistical algorithms that can learn from data and generate insights or predictions. It involves creating models that can automatically learn and improve from experience without being explicitly programmed. Machine learning is used in various applications such as image recognition, natural language processing, and recommendation systems.'

In [None]:
query2 = agent_executer.invoke({"input": "what's the paper 2508.09035 "})
query2.get('output')



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `arxiv` with `{'query': '2508.09035'}`


[0m[33;1m[1;3mPublished: 2025-08-12
Title: P/D-Device: Disaggregated Large Language Model between Cloud and Devices
Authors: Yibo Jin, Yixu Xu, Yue Chen, Chengbin Wang, Tao Wang, Jiaqi Huang, Rongfei Zhang, Yiming [0m[32;1m[1;3mThe paper with the identifier 2508.09035 is titled "P/D-Device: Disaggregated Large Language Model between Cloud and Devices" authored by Yibo Jin, Yixu Xu, Yue Chen, Chengbin Wang, Tao Wang, Jiaqi Huang, Rongfei Zhang, and Yiming. It was published on August 12, 2025.[0m

[1m> Finished chain.[0m


'The paper with the identifier 2508.09035 is titled "P/D-Device: Disaggregated Large Language Model between Cloud and Devices" authored by Yibo Jin, Yixu Xu, Yue Chen, Chengbin Wang, Tao Wang, Jiaqi Huang, Rongfei Zhang, and Yiming. It was published on August 12, 2025.'