# Building a research Agent with LangChain

In [1]:
# !pip install openai
# !pip install "langchain[all]"

In [2]:
from dotenv import load_dotenv

load_dotenv()

# Or if you are in Colab, uncoment below and add your api key
import os
# os.environ["OPENAI_API_KEY"] = "your-api-key"

In [3]:
from langchain_openai.chat_models import ChatOpenAI

In [4]:
llm_chat = ChatOpenAI(temperature=0)

In [5]:
from langchain.utilities import SerpAPIWrapper
from langchain.agents import tool

search = SerpAPIWrapper()

@tool
def web_search(query: str):
    """Searches the web given a user query and returns the most relevant links"""
    output = search.results(query)
    results = output["organic_results"]
    return [r["link"] for r in results]

web_search("Current resources on why pancakes are the best breakfast")

['https://www.reddit.com/r/AskCulinary/comments/emeul2/why_are_diner_style_pancakes_so_good_regardless/',
 'https://www.quora.com/Why-do-pancakes-at-restaurants-like-Dennys-or-IHOP-taste-better-than-homemade-ones',
 'https://www.prnewswire.com/news-releases/packaged-facts-still-room-for-pancakes-and-waffles-at-americas-breakfast-table-300317750.html',
 'https://www.fitday.com/fitness-articles/nutrition/healthy-eating/healthy-breakfast-pancakes-vs-waffles.html',
 'https://www.buzzfeed.com/ryankincaid/stacks-on-stacks',
 'https://www.snackandbakery.com/articles/94837-state-of-the-industry-2020-breakfast-products-combine-nutrition-and-good-taste',
 'https://health.ucdavis.edu/blog/good-food/why-is-breakfast-important-our-dietitian-gives-healthy-breakfast-ideas/2022/04',
 'https://www.kingarthurbaking.com/blog/2021/10/26/best-pancake-tips',
 'https://www.smithsonianmag.com/history/a-brief-history-of-pancakes-180981667/']

In [6]:
from langchain import hub

prompt = hub.pull("hwchase17/openai-functions-agent")

prompt

ChatPromptTemplate(input_variables=['agent_scratchpad', 'input'], input_types={'chat_history': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]], 'agent_scratchpad': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]]}, messages=[SystemMessagePromptTemplate(prompt=PromptTemplate(input_variables=[], template='You are a helpful assistant')), MessagesPlaceholder(variable_name='chat_history', optional=True), HumanMessagePromptTemplate(prompt=PromptTemplate(input_variables=['input'], template='{input}')), MessagesPlaceholder(

In [7]:
prompt.input_variables

['agent_scratchpad', 'input']

In [8]:
prompt.input_types

{'chat_history': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]],
 'agent_scratchpad': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]]}

In [9]:
prompt.__dict__

{'name': None,
 'input_variables': ['agent_scratchpad', 'input'],
 'input_types': {'chat_history': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]],
  'agent_scratchpad': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]]},
 'output_parser': None,
 'partial_variables': {},
 'metadata': None,
 'tags': None,
 'messages': [SystemMessagePromptTemplate(prompt=PromptTemplate(input_variables=[], template='You are a helpful assistant')),
  MessagesPlaceholder(variable_name='chat_history', optional=True),
  HumanMessagePromptTe

To build an agent I need to do 2 essential things:

1. Connect a model to tools
2. Set up a loop in which the model continuously tries to solve the problem by either just calling the llm or calling a tool, observing the result and repeating until finished.

In [10]:
prompt

ChatPromptTemplate(input_variables=['agent_scratchpad', 'input'], input_types={'chat_history': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]], 'agent_scratchpad': typing.List[typing.Union[langchain_core.messages.ai.AIMessage, langchain_core.messages.human.HumanMessage, langchain_core.messages.chat.ChatMessage, langchain_core.messages.system.SystemMessage, langchain_core.messages.function.FunctionMessage, langchain_core.messages.tool.ToolMessage]]}, messages=[SystemMessagePromptTemplate(prompt=PromptTemplate(input_variables=[], template='You are a helpful assistant')), MessagesPlaceholder(variable_name='chat_history', optional=True), HumanMessagePromptTemplate(prompt=PromptTemplate(input_variables=['input'], template='{input}')), MessagesPlaceholder(

In [11]:
from langchain.agents import create_openai_tools_agent

tools = [web_search]

agent = create_openai_tools_agent(llm_chat, tools, prompt)

In [12]:
from langchain.agents import AgentExecutor


agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True)

In [13]:
agent_executor.invoke({"input": "What are good resources to learn about learning new languages?",
                             "agent_scratchpad": []})



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `web_search` with `{'query': 'resources to learn new languages'}`


[0m[36;1m[1;3m['https://nymag.com/strategist/article/how-to-learn-languages-at-home.html', 'https://www.lingualift.com/blog/best-language-learning-resources/', 'https://www.theintrepidguide.com/top-language-learning-resources/', 'https://www.reddit.com/r/languagelearning/comments/uhiuab/what_are_apps_or_websites_where_you_can_learn/', 'https://www.quora.com/What-resources-do-you-use-to-learn-languages', 'https://libguides.csudh.edu/modernlanguages/freeresources', 'https://www.fluentu.com/blog/free-language-learning-websites/', 'https://www.alllanguageresources.com/'][0m[32;1m[1;3mI found some resources that can help you learn new languages:

1. [How to Learn Languages at Home - NYMag](https://nymag.com/strategist/article/how-to-learn-languages-at-home.html)
2. [Best Language Learning Resources - LinguaLift](https://www.lingualift.com/blog/bes

{'input': 'What are good resources to learn about learning new languages?',
 'agent_scratchpad': [],
 'output': 'I found some resources that can help you learn new languages:\n\n1. [How to Learn Languages at Home - NYMag](https://nymag.com/strategist/article/how-to-learn-languages-at-home.html)\n2. [Best Language Learning Resources - LinguaLift](https://www.lingualift.com/blog/best-language-learning-resources/)\n3. [Top Language Learning Resources - The Intrepid Guide](https://www.theintrepidguide.com/top-language-learning-resources/)\n4. [Apps or Websites to Learn Languages - Reddit](https://www.reddit.com/r/languagelearning/comments/uhiuab/what_are_apps_or_websites_where_you_can_learn/)\n5. [Resources for Learning Languages - Quora](https://www.quora.com/What-resources-do-you-use-to-learn-languages)\n6. [Free Language Learning Websites - FluentU](https://www.fluentu.com/blog/free-language-learning-websites/)\n7. [Language Learning Resources - All Language Resources](https://www.allla

In [15]:
from langchain.agents import AgentExecutor, create_react_agent

prompt = hub.pull("hwchase17/react")

agent = create_react_agent(llm_chat, tools, prompt)

react_agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True)


react_agent_executor.invoke({"input": "What is the best way to learn good programming practices?",
                             "agent_scratchpad": []})



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3mLearning good programming practices is essential for becoming a successful programmer.
Action: web_search
Action Input: "best way to learn good programming practices"[0m[36;1m[1;3m['https://www.reddit.com/r/learnprogramming/comments/6bv3ux/how_do_you_practice_programming/', 'https://www.quora.com/How-can-I-learn-the-best-practices-in-programming', 'https://www.thinkful.com/blog/coding-best-practices/', 'https://www.educative.io/blog/coding-best-practices', 'https://www.datacamp.com/tutorial/coding-best-practices-and-guidelines', 'https://flatironschool.com/blog/best-websites-to-practice-coding-for-beginners/', 'https://codesigningstore.com/ultimate-programming-best-practices-guide', 'https://www.browserstack.com/guide/coding-standards-best-practices', 'https://medium.com/@josueparra2892/20-best-programming-practices-407df688b96e'][0m[32;1m[1;3mThere are many resources available online to learn good programming practices

{'input': 'What is the best way to learn good programming practices?',
 'agent_scratchpad': [],
 'output': 'The best way to learn good programming practices is to practice coding regularly on websites like LeetCode, CodeWars, and Codecademy.'}

Now let's build this same agent but using a more in depth approach going one level down in the abstraction layers.

Now, let's go down the latter of abstraction and build this same agent but leveraging the lower-level capabilities of LangChain's LCEL language. 

In [17]:
from langchain_openai.chat_models import ChatOpenAI

In [18]:
llm_chat = ChatOpenAI(model="gpt-3.5-turbo-16k",temperature=0)

In [19]:
from langchain.agents import tool
from langchain.utilities import SerpAPIWrapper

search = SerpAPIWrapper()

@tool
def web_search(query: str):
    """Searches the web given a user query and returns the most relevant links"""
    output = search.results(query)
    results = output["organic_results"]
    return [r["link"] for r in results]


tools = [web_search]

In [20]:
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder

In [21]:
prompt = ChatPromptTemplate.from_messages(
    [
        (
            "system", "You are a powerful research assistant.\
            Given a user query you will perform research on the \
                topic using the tools available."
        ),
        ("user", "{input}"),
        MessagesPlaceholder(variable_name="agent_scratchpad"),
    ]
)

# Bind tools to the LLM

In [22]:
from langchain_community.tools.convert_to_openai import format_tool_to_openai_tool

In [23]:
llm_with_tools = llm_chat.bind(tools=[format_tool_to_openai_tool(tool) for tool in tools])

  warn_deprecated(


Now, all we have to do is import two last utility functions:
- a component for formatting intermediate steps (agent action, tool output pairs) to input messages that can be sent to the model, 
- and a component for converting the output message into an agent action/agent finish.

In [24]:
from langchain.agents.format_scratchpad.openai_tools import format_to_openai_tool_messages
from langchain.agents.output_parsers.openai_tools import OpenAIToolsAgentOutputParser

In [25]:
agent = (
    {
        "input": lambda x: x["input"], 
        "agent_scratchpad": lambda x: format_to_openai_tool_messages(
            x["intermediate_steps"]
        ),
    }
    | prompt
    | llm_with_tools
    | OpenAIToolsAgentOutputParser()
)

In [26]:
from langchain.agents import AgentExecutor

agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True)

In [27]:
list(agent_executor.stream({"input": "What are the latest open source LLMs available?"}))



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `web_search` with `{'query': 'latest open source LLMs'}`


[0m[36;1m[1;3m['https://github.com/eugeneyan/open-llms', 'https://www.datacamp.com/blog/top-open-source-llms', 'https://www.unite.ai/best-open-source-llms/', 'https://www.elastic.co/blog/open-source-llms-guide', 'https://blog.n8n.io/open-source-llm/', 'https://klu.ai/blog/open-source-llm-models', 'https://github.com/Hannibal046/Awesome-LLM', 'https://www.simform.com/blog/open-source-llm/', 'https://www.techopedia.com/6-best-open-source-llms-to-watch-out-for-in-2024', 'https://www.upstage.ai/feed/insight/top-open-source-llms-2024'][0m[32;1m[1;3mHere are some of the latest open source LLMs available:

1. [Open-LLMs](https://github.com/eugeneyan/open-llms) - A collection of open source LLM models on GitHub.
2. [Top Open Source LLMs](https://www.datacamp.com/blog/top-open-source-llms) - A blog post on DataCamp that lists some of the top open source LLMs.


[{'actions': [OpenAIToolAgentAction(tool='web_search', tool_input={'query': 'latest open source LLMs'}, log="\nInvoking: `web_search` with `{'query': 'latest open source LLMs'}`\n\n\n", message_log=[AIMessageChunk(content='', additional_kwargs={'tool_calls': [{'index': 0, 'id': 'call_phz3F3TMThgLMYTlLtuLDKN0', 'function': {'arguments': '{\n  "query": "latest open source LLMs"\n}', 'name': 'web_search'}, 'type': 'function'}]})], tool_call_id='call_phz3F3TMThgLMYTlLtuLDKN0')],
  'messages': [AIMessageChunk(content='', additional_kwargs={'tool_calls': [{'index': 0, 'id': 'call_phz3F3TMThgLMYTlLtuLDKN0', 'function': {'arguments': '{\n  "query": "latest open source LLMs"\n}', 'name': 'web_search'}, 'type': 'function'}]})]},
 {'steps': [AgentStep(action=OpenAIToolAgentAction(tool='web_search', tool_input={'query': 'latest open source LLMs'}, log="\nInvoking: `web_search` with `{'query': 'latest open source LLMs'}`\n\n\n", message_log=[AIMessageChunk(content='', additional_kwargs={'tool_calls

Now, what if we want the agent to actually read the links and return a research report rather than just the urls?

In [28]:
import requests
from bs4 import BeautifulSoup


@tool
def scrape_text(url: str):
    """Scrapes the text from a given URL and returns it."""
    # Send a GET request to the webpage
    try:
        response = requests.get(url)

        # Check if the request was successful
        if response.status_code == 200:
            # Parse the content of the request with BeautifulSoup
            soup = BeautifulSoup(response.text, "html.parser")

            # Extract all text from the webpage
            page_text = soup.get_text(separator=" ", strip=True)

            # Print the extracted text
            return page_text[:10000]
        else:
            return f"Failed to retrieve the webpage: Status code {response.status_code}"
    except Exception as e:
        print(e)
        return f"Failed to retrieve the webpage: {e}"

In [29]:
tools = [web_search, scrape_text]

In [30]:
llm_with_tools = llm_chat.bind(tools=[format_tool_to_openai_tool(tool) for tool in tools])

In [31]:
prompt = ChatPromptTemplate.from_messages(
    [
        (
            "system", "You are a powerful research assistant. Given a user query you will perform research on the topic using the tools available."
        ),
        ("user", "Scrape and summarize the contents from these urls: {input}"),
        MessagesPlaceholder(variable_name="agent_scratchpad"),
    ]
)

In [32]:
agent = (
    {
        "input": lambda x: x["input"], 
        "agent_scratchpad": lambda x: format_to_openai_tool_messages(
            x["intermediate_steps"]
        ),
    }
    | prompt
    | llm_with_tools
    | OpenAIToolsAgentOutputParser()
)

In [33]:
from langchain.agents import AgentExecutor

agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True)

In [34]:
agent_executor.invoke({"input": "What are the current best practices to deploy LLM agents to production?"})



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `scrape_text` with `{'url': 'https://towardsdatascience.com/deploying-llm-agents-to-production-best-practices-7e4e8e8e8e8e'}`


[0m[33;1m[1;3mFailed to retrieve the webpage: Status code 403[0m[32;1m[1;3m
Invoking: `scrape_text` with `{'url': 'https://towardsdatascience.com/deploying-llm-agents-to-production-best-practices-7e4e8e8e8e8e'}`


[0m[33;1m[1;3mFailed to retrieve the webpage: Status code 403[0m[32;1m[1;3m
Invoking: `scrape_text` with `{'url': 'https://towardsdatascience.com/deploying-llm-agents-to-production-best-practices-7e4e8e8e8e8e'}`


[0m[33;1m[1;3mFailed to retrieve the webpage: Status code 403[0m[32;1m[1;3m
Invoking: `web_search` with `{'query': 'best practices to deploy LLM agents to production'}`


[0m[36;1m[1;3m['https://www.qwak.com/post/best-practices-when-deploying-llms-in-production', 'https://arize.com/blog-course/large-language-model-llm-deployment/', 'https://www.redd

{'input': 'What are the current best practices to deploy LLM agents to production?',
 'output': 'I apologize, but I am unable to access the content from the provided URL. However, based on my research, here are some general best practices for deploying LLM agents to production:\n\n1. Infrastructure: Ensure that you have a robust and scalable infrastructure to handle the computational requirements of LLM agents. This may involve using cloud-based services or dedicated hardware.\n\n2. Model Training: Train your LLM agents on a diverse and representative dataset to improve their performance and generalization capabilities. Consider using transfer learning techniques to leverage pre-trained models.\n\n3. Fine-tuning: Fine-tune your LLM agents on specific tasks or domains to improve their performance and adapt them to your specific use case. This may involve using task-specific datasets and optimizing hyperparameters.\n\n4. Prompt Engineering: Craft clear and specific prompts to guide the o

We encounter a few issues when we try to just add the scrape function to the agent without any other consideration. 
How about we make this better by improving the inner chain of the agent?

In [35]:
from langchain.prompts import MessagesPlaceholder

MEMORY_KEY = "chat_history"
prompt = ChatPromptTemplate.from_messages(
    [
        (
            "system",
            "You are very powerful assistant, but bad at calculating lengths of words.",
        ),
        MessagesPlaceholder(variable_name=MEMORY_KEY),
        ("user", "{input}"),
        MessagesPlaceholder(variable_name="agent_scratchpad"),
    ]
)

from langchain_core.messages import AIMessage, HumanMessage

chat_history = []

agent = (
    {
        "input": lambda x: x["input"],
        "agent_scratchpad": lambda x: format_to_openai_tool_messages(
            x["intermediate_steps"]
        ),
        "chat_history": lambda x: x["chat_history"],
    }
    | prompt
    | llm_with_tools
    | OpenAIToolsAgentOutputParser()
)
agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True)

input1 = "how many letters in the word educa?"
result = agent_executor.invoke({"input": input1, "chat_history": chat_history})
chat_history.extend(
    [
        HumanMessage(content=input1),
        AIMessage(content=result["output"]),
    ]
)
agent_executor.invoke({"input": "is that a real word?", "chat_history": chat_history})



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3mThe word "educa" has 5 letters.[0m

[1m> Finished chain.[0m


[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m
Invoking: `web_search` with `{'query': 'define educa'}`


[0m[36;1m[1;3m['https://en.wikipedia.org/wiki/Education', 'https://www.merriam-webster.com/dictionary/education', 'https://www.dictionary.com/browse/education', 'https://www.britannica.com/topic/education', 'https://dictionary.cambridge.org/us/dictionary/english/education', 'https://learn.org/articles/What_is_Education.html', 'https://leverageedu.com/blog/what-is-education/', 'https://kids.wordsmyth.net/we/?rid=13075&ent_l=education', 'https://link.springer.com/chapter/10.1007/978-3-030-83925-3_3', 'https://examplanning.com/definition-of-education-by-different-authors/'][0m[32;1m[1;3mYes, "educa" is not a recognized word in the English language. It may be a misspelling or a variation of the word "education".[0m

[1m> Finished chain.[0m


{'input': 'is that a real word?',
 'chat_history': [HumanMessage(content='how many letters in the word educa?'),
  AIMessage(content='The word "educa" has 5 letters.')],
 'output': 'Yes, "educa" is not a recognized word in the English language. It may be a misspelling or a variation of the word "education".'}