In [None]:
from langchain_huggingface import ChatHuggingFace, HuggingFacePipeline

llm = HuggingFacePipeline.from_model_id(
    model_id="unsloth/DeepSeek-R1-Distill-Qwen-14B-unsloth-bnb-4bit",
    task="text-generation",
    pipeline_kwargs=dict(
        max_new_tokens=512,
        do_sample=False,
        repetition_penalty=1.03,
    ),
)

chat_model = ChatHuggingFace(llm=llm)

  from .autonotebook import tqdm as notebook_tqdm
`low_cpu_mem_usage` was None, now default to True since model is quantized.
Downloading shards:   0%|                                         | 0/3 [00:00<?, ?it/s]

In [None]:
from langchain_core.messages import HumanMessage, SystemMessage

messages = [
    SystemMessage("Tell me something about yourself"),
    HumanMessage("hi!"),
]

chat_model.invoke(messages)

In [None]:
from typing import List, Optional
from pydantic import BaseModel, Field


class Person(BaseModel):
    """Information about a person."""

    # ^ Doc-string for the entity Person.
    # This doc-string is sent to the LLM as the description of the schema Person,
    # and it can help to improve extraction results.

    # Note that:
    # 1. Each field is an `optional` -- this allows the model to decline to extract it!
    # 2. Each field has a `description` -- this description is used by the LLM.
    # Having a good description can help improve extraction results.
    name: Optional[str] = Field(default=None, description="The name of the person")
    hair_color: Optional[str] = Field(
        default=None, description="The color of the person's hair if known"
    )
    height_in_meters: Optional[str] = Field(
        default=None, description="Height measured in meters"
    )

class Data(BaseModel):
    """Extracted data about people."""

    # Creates a model so that we can extract multiple entities.
    people: List[Person]

In [None]:
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder

prompt_template = ChatPromptTemplate.from_messages(
    [
        (
            "system",
            "You are an expert extraction algorithm. "
            "Only extract relevant information from the text. "
            "If you do not know the value of an attribute asked to extract, "
            "return null for the attribute's value.",
        ),
        # Please see the how-to about improving performance with
        # reference examples.
        # MessagesPlaceholder('examples'),
        ("human", "{text}"),
    ]
)

In [None]:
text = "Alan Smith is 6 feet tall and has blond hair."
prompt = prompt_template.invoke({"text": text})
chat_model.invoke(prompt)

In [15]:
text = "My name is Jeff, my hair is black and i am 6 feet tall. Anna has the same color hair as me."
prompt = prompt_template.invoke({"text": text})
chat_model.invoke(prompt)



AIMessage(content="<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nCutting Knowledge Date: December 2023\nToday Date: 26 Jul 2024\n\nYou are an expert extraction algorithm. Only extract relevant information from the text. If you do not know the value of an attribute asked to extract, return null for the attribute's value.<|eot_id|><|start_header_id|>user<|end_header_id|>\n\nMy name is Jeff, my hair is black and i am 6 feet tall. Anna has the same color hair as me.<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\nHere's the extracted information:\n\n- Name: Jeff\n- Hair Color: black\n- Height: 6 feet\n- Name of other person with the same hair color: Anna", additional_kwargs={}, response_metadata={}, id='run-ae1fa1ff-74dc-454a-b936-f6beeb3f502d-0')

In [16]:
from langchain_core.utils.function_calling import tool_example_to_messages

examples = [
    (
        "The ocean is vast and blue. It's more than 20,000 feet deep.",
        Data(people=[]),
    ),
    (
        "Fiona traveled far from France to Spain.",
        Data(people=[Person(name="Fiona", height_in_meters=None, hair_color=None)]),
    ),
]


messages = []

for txt, tool_call in examples:
    if tool_call.people:
        # This final message is optional for some providers
        ai_response = "Detected people."
    else:
        ai_response = "Detected no people."
    messages.extend(tool_example_to_messages(txt, [tool_call], ai_response=ai_response))

  messages.extend(tool_example_to_messages(txt, [tool_call], ai_response=ai_response))


In [17]:
for message in messages:
    message.pretty_print()


The ocean is vast and blue. It's more than 20,000 feet deep.
Tool Calls:
  Data (97c59f6d-436f-4cac-8e74-67f201287df0)
 Call ID: 97c59f6d-436f-4cac-8e74-67f201287df0
  Args:
    people: []

You have correctly called this tool.

Detected no people.

Fiona traveled far from France to Spain.
Tool Calls:
  Data (2d52cfbd-8272-4c92-98f3-3f7e758b3d5c)
 Call ID: 2d52cfbd-8272-4c92-98f3-3f7e758b3d5c
  Args:
    people: [{'name': 'Fiona', 'hair_color': None, 'height_in_meters': None}]

You have correctly called this tool.

Detected people.


In [18]:

chat_model.invoke([HumanMessage(content="Hi! I'm Bob")])



AIMessage(content="<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nCutting Knowledge Date: December 2023\nToday Date: 26 Jul 2024\n\n<|eot_id|><|start_header_id|>user<|end_header_id|>\n\nHi! I'm Bob<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\nHello Bob. What's on your mind?", additional_kwargs={}, response_metadata={}, id='run-853782b2-fc9c-4cc7-9fc5-ba088fafe75a-0')

In [20]:
chat_model.invoke([HumanMessage(content="What's my name?")])



AIMessage(content="<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nCutting Knowledge Date: December 2023\nToday Date: 26 Jul 2024\n\n<|eot_id|><|start_header_id|>user<|end_header_id|>\n\nWhat's my name?<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\nI don't have any information about your personal details, including your name. This is the beginning of our conversation, and I'm happy to chat with you. How can I assist you today?", additional_kwargs={}, response_metadata={}, id='run-e50dcd6c-3d49-4720-a8f9-d5c23aa64d7c-0')

In [55]:
from langgraph.checkpoint.memory import MemorySaver
from langgraph.graph import START, MessagesState, StateGraph

# Define a new graph
workflow = StateGraph(state_schema=MessagesState)


# Define the function that calls the model
def call_model(state: MessagesState):
    response = chat_model.invoke(state["messages"])
    return {"messages": response}


# Define the (single) node in the graph
workflow.add_edge(START, "model")
workflow.add_node("model", call_model)

# Add memory
memory = MemorySaver()
app = workflow.compile(checkpointer=memory)

In [56]:
config = {"configurable": {"thread_id": "abc123"}}

In [57]:
query = "Hi! I'm Bob."

input_messages = [HumanMessage(query)]
output = app.invoke({"messages": input_messages}, config)
output["messages"][-1].pretty_print()

You seem to be using the pipelines sequentially on GPU. In order to maximize efficiency please use a dataset



<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

Hi! I'm Bob.<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Hello Bob! It's nice to meet you. Is there something I can help you with or would you like to chat?


In [58]:
input_messages = [HumanMessage("What's my name")]
output = app.invoke({"messages": input_messages}, config)
output["messages"][-1].pretty_print()




<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

Hi! I'm Bob.<|eot_id|><|start_header_id|>assistant<|end_header_id|>

<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

Hi! I'm Bob.<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Hello Bob! It's nice to meet you. Is there something I can help you with or would you like to chat?<|eot_id|><|start_header_id|>user<|end_header_id|>

What's my name<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Your name is Bob.


In [60]:
from langchain_core.messages import AIMessage

config = {"configurable": {"thread_id": "abc789"}}
query = "Hi I'm Todd, please tell me a joke."
language = "English"

input_messages = [HumanMessage(query)]
for chunk, metadata in app.stream(
    {"messages": query, "language": language},
    config,
    stream_mode="messages",
):
    if isinstance(chunk, AIMessage):  # Filter to just model responses
        print(chunk.content, end="|")



<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

Hi I'm Todd, please tell me a joke.<|eot_id|><|start_header_id|>user<|end_header_id|>

Hi I'm Todd, please tell me a joke.<|eot_id|><|start_header_id|>assistant<|end_header_id|>

A man walked into a library and asked the librarian, "Do you have any books on Pavlov's dogs and Schrödinger's cat?" The librarian replied, "It rings a bell, but I'm not sure if it's here or not."|

In [69]:
from langchain_community.tools.ddg_search.tool import DuckDuckGoSearchRun
from langchain_core.messages import HumanMessage
from langgraph.checkpoint.memory import MemorySaver
from langgraph.prebuilt import create_react_agent

In [70]:
memory = MemorySaver()
search = DuckDuckGoSearchRun(max_results=2)
tools = [search]
agent_executor = create_react_agent(chat_model, tools, checkpointer=memory)

In [71]:
config = {"configurable": {"thread_id": "abc123"}}
for step in agent_executor.stream(
    {"messages": [HumanMessage(content="hi im bob! and i live in sf")]},
    config,
    stream_mode="values",
):
    step["messages"][-1].pretty_print()


hi im bob! and i live in sf





<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

hi im bob! and i live in sf<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Hi Bob! It's great to meet you. San Francisco is a fantastic city, isn't it? What do you like most about living there? The Golden Gate Bridge, the food scene, or something else?


In [72]:
for step in agent_executor.stream(
    {"messages": [HumanMessage(content="whats the weather where I live?")]},
    config,
    stream_mode="values",
):
    step["messages"][-1].pretty_print()


whats the weather where I live?





<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

hi im bob! and i live in sf<|eot_id|><|start_header_id|>assistant<|end_header_id|>

<|begin_of_text|><|start_header_id|>system<|end_header_id|>

Cutting Knowledge Date: December 2023
Today Date: 26 Jul 2024

<|eot_id|><|start_header_id|>user<|end_header_id|>

hi im bob! and i live in sf<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Hi Bob! It's great to meet you. San Francisco is a fantastic city, isn't it? What do you like most about living there? The Golden Gate Bridge, the food scene, or something else?<|eot_id|><|start_header_id|>user<|end_header_id|>

whats the weather where I live?<|eot_id|><|start_header_id|>assistant<|end_header_id|>

San Francisco's weather is pretty unique. As of my knowledge cutoff in December 2023, San Francisco is known for its cool and foggy climate, especially during the