In [1]:
import os
from dotenv import load_dotenv

load_dotenv()

True

In [2]:
!ollama list

NAME                       ID              SIZE      MODIFIED     
nomic-embed-text:latest    0a109f422b47    274 MB    2 weeks ago     
phi3:latest                4f2222927938    2.2 GB    2 months ago    
gemma2:2b                  8ccf136fdd52    1.6 GB    2 months ago    
llama3.2:latest            a80c4f17acd5    2.0 GB    2 months ago    
deepseek-r1:8b             28f8fd6cdc67    4.9 GB    2 months ago    


In [3]:
EMBEDDING_MODEL = "nomic-embed-text"
LLM_MODEL = "llama3.2"

In [16]:
from langchain_ollama import ChatOllama
from langchain_ollama import OllamaEmbeddings
from langchain_core.prompts import ChatPromptTemplate
from langchain_core.output_parsers import StrOutputParser
from langchain_core.runnables import RunnablePassthrough, RunnableLambda


llm = ChatOllama(model=LLM_MODEL, temperature=0)

embedding = OllamaEmbeddings(model=EMBEDDING_MODEL)

In [None]:
# Load blog
import bs4
from langchain_community.document_loaders import WebBaseLoader
loader = WebBaseLoader(
    web_paths=("https://lilianweng.github.io/posts/2023-06-23-agent/",),
    bs_kwargs=dict(
        parse_only=bs4.SoupStrainer(
            class_=("post-content", "post-title", "post-header")
        )
    ),
)
blog_docs = loader.load()

In [31]:
# Split
from langchain.text_splitter import RecursiveCharacterTextSplitter
text_splitter = RecursiveCharacterTextSplitter.from_tiktoken_encoder(chunk_size=300, chunk_overlap=50)

splits = text_splitter.split_documents(blog_docs)

In [5]:
import chromadb

chroma_client = chromadb.Client()
collections = chroma_client.list_collections()

collections

# chroma_client = delete_collection("langchain")
# chroma_client.list_collections()

[]

Uncomment and run this first if it's your first time indexing these documents

In [6]:
# from langchain_community.vectorstores import Chroma

# # Embed
# vectorstore = Chroma.from_documents(
#     persist_directory="./chroma_db",
#     documents=splits,
#     embedding=OllamaEmbeddings(
#         model=EMBEDDING_MODEL
#     )
# )

# retriever = vectorstore.as_retriever()

In [7]:
from langchain_chroma import Chroma

# Embed
vectorstore = Chroma(
    persist_directory="./chroma_db",
    embedding_function=embedding
)

retriever = vectorstore.as_retriever()

# Query Translation

Distance-based vector database retrieval embeds queries in high-dimensional space and finds similar embedded documents based on a distance metric.

But, retrieval may produce different results with subtle changes in query wording, or if the embeddings do not capture the semantics of the data well. Prompt engineering / tuning is sometimes done to manually address these problems, but can be tedious.

## Multi Query Retrieval

Using an LLM to generate multiple queries from different perspectives for a given user input query.

![image.png](attachment:feaa3bf3-e447-45ec-ab60-2f47daf6478e.png)

### Prompt

In [46]:
from langchain.prompts import ChatPromptTemplate

# Multi Query: Different Perspectives
template = """You are an AI language model assistant. Your task is to generate exactly five 
different versions of the given user question to retrieve relevant documents from a vector 
database. Provide these alternative questions as a plain list without any introductory text 
or numbering. Original question: {question}"""
prompt_perspectives = ChatPromptTemplate.from_template(template)

In [47]:
from langchain_core.output_parsers import StrOutputParser

generate_queries = (
    prompt_perspectives 
    | llm
    | StrOutputParser() 
    | (lambda x: x.split("\n"))
)

In [48]:
question = "What is task decomposition for LLM agents?"
docs = generate_queries.invoke({"question":question})
len(docs)

5

In [49]:
docs

['What is the concept of task decomposition in Large Language Model (LLM) agents?',
 'How does task decomposition work for LLM agents?',
 'What are the benefits of using task decomposition with LLM agents?',
 'Can you explain the process of task decomposition for LLM agents?',
 'How does task decomposition improve the performance of LLM agents?']

In [50]:
from langchain.load import dumps, loads

def get_unique_union(documents: list[list]):
    """ Unique union of retrieved docs """
    # Flatten list of lists, and convert each Document to string
    flattened_docs = [dumps(doc) for sublist in documents for doc in sublist]
    # Get unique documents
    unique_docs = list(set(flattened_docs))
    # Return
    return [loads(doc) for doc in unique_docs]

# Retrieve
question = "What is task decomposition for LLM agents?"
retrieval_chain = generate_queries | retriever.map() | get_unique_union
docs = retrieval_chain.invoke({"question":question})
len(docs)

5

![image.png](attachment:dc09abee-73c1-4f6e-888a-92ec87e75ca1.png)

In [51]:
from operator import itemgetter
from langchain_ollama import ChatOllama
from langchain_core.runnables import RunnablePassthrough

# RAG
template = """Answer the following question based on this context:

{context}

Question: {question}
"""

prompt = ChatPromptTemplate.from_template(template)

final_rag_chain = (
    {"context": retrieval_chain, 
     "question": itemgetter("question")} 
    | prompt
    | llm
    | StrOutputParser()
)

final_rag_chain.invoke({"question":question})

'According to the text, task decomposition for LLM (Large Language Model) agents involves breaking down large tasks into smaller, manageable subgoals. This can be done in three ways:\n\n1. Using simple prompting, such as "Steps for XYZ. 1.", or "What are the subgoals for achieving XYZ?"\n2. Using task-specific instructions, such as "Write a story outline." for writing a novel\n3. With human inputs.\n\nTask decomposition is used to enable efficient handling of complex tasks and allows the LLM agent to plan ahead and make progress towards its goals.'

![image.png](attachment:25c97949-5879-49a3-a323-f3703495971c.png)

## Rag Fusion

Using an LLM to generate multiple queries based on a signle input query, then using reciprocal rank fusion that takes mutiple lists of ranked documents and aggregates them to a final output ranking.

### Prompt

In [8]:
from langchain.prompts import ChatPromptTemplate

# RAG-Fusion: Related
template = """You are a helpful assistant that generates multiple search queries based on a single input query. \n
Provide these alternative questions as a plain list without any introductory text 
or numbering. Generate multiple search queries related to: {question} \n
Output (4 queries):"""
prompt_rag_fusion = ChatPromptTemplate.from_template(template)

In [9]:
from langchain_core.output_parsers import StrOutputParser

generate_queries = (
    prompt_rag_fusion 
    | llm
    | StrOutputParser() 
    | (lambda x: x.split("\n"))
)

In [10]:
question = "What is task decomposition for LLM agents?"
docs = generate_queries.invoke({"question":question})
len(docs)

4

In [11]:
docs

['What is the purpose of task decomposition in large language model (LLM) agents?',
 'How does task decomposition improve the performance of LLM agents?',
 'What are the benefits of using task decomposition for LLM agents?',
 'Can task decomposition be used to address common challenges in LLM agent development?']

In [12]:
from langchain.load import dumps, loads

def reciprocal_rank_fusion(results: list[list], k=60):
    """ Reciprocal_rank_fusion that takes multiple lists of ranked documents 
        and an optional parameter k used in the RRF formula """
    
    # Initialize a dictionary to hold fused scores for each unique document
    fused_scores = {}

    # Iterate through each list of ranked documents
    for docs in results:
        # Iterate through each document in the list, with its rank (position in the list)
        for rank, doc in enumerate(docs):
            # Convert the document to a string format to use as a key (assumes documents can be serialized to JSON)
            doc_str = dumps(doc)
            # If the document is not yet in the fused_scores dictionary, add it with an initial score of 0
            if doc_str not in fused_scores:
                fused_scores[doc_str] = 0
            # Retrieve the current score of the document, if any
            previous_score = fused_scores[doc_str]
            # Update the score of the document using the RRF formula: 1 / (rank + k)
            fused_scores[doc_str] += 1 / (rank + k)

    # Sort the documents based on their fused scores in descending order to get the final reranked results
    reranked_results = [
        (loads(doc), score)
        for doc, score in sorted(fused_scores.items(), key=lambda x: x[1], reverse=True)
    ]

    # Return the reranked results as a list of tuples, each containing the document and its fused score
    return reranked_results

In [13]:
# Retrieve
retrieval_chain_rag_fusion = generate_queries | retriever.map() | reciprocal_rank_fusion
docs = retrieval_chain_rag_fusion.invoke({"question": question})
len(docs)

  (loads(doc), score)


5

In [14]:
from operator import itemgetter

# RAG
template = """Answer the following question based on this context:

{context}

Question: {question}
"""

prompt = ChatPromptTemplate.from_template(template)

final_rag_chain = (
    {"context": retrieval_chain_rag_fusion, 
     "question": itemgetter("question")} 
    | prompt
    | llm
    | StrOutputParser()
)

final_rag_chain.invoke({"question":question})

'Task decomposition for LLM (Large Language Model) agents involves breaking down large tasks into smaller, manageable subgoals. This allows the agent to efficiently handle complex tasks.\n\nThere are three ways to perform task decomposition:\n\n1. Using simple prompting with instructions like "Steps for XYZ. 1.", "What are the subgoals for achieving XYZ?"\n2. Using task-specific instructions, such as "Write a story outline." for writing a novel\n3. With human inputs\n\nTask decomposition enables efficient handling of complex tasks and is an important component of LLM-powered autonomous agent systems.'

![image.png](attachment:e9103b42-1d95-42a6-9174-6ba274e8cf2d.png)

## Decomposition

Using an LLM to generate multiple sub-questions related to a single input question, then dynamically retrieve to help solve the sub-problems.

### Prompt

In [19]:
from langchain.prompts import ChatPromptTemplate

# Decomposition
template = """You are a helpful assistant that generates multiple sub-questions related to an input question. \n
The goal is to break down the input into a set of sub-problems / sub-questions that can be answers in isolation. \n
Provide these sub-questions as a plain list without any introductory text 
or numbering.
Generate multiple search queries related to: {question} \n
Output (3 queries):"""
prompt_decomposition = ChatPromptTemplate.from_template(template)

In [37]:
from langchain_core.output_parsers import StrOutputParser

generate_queries_decomposition = (
    prompt_decomposition 
    | llm
    | StrOutputParser() 
    | (lambda x: x.split("\n"))
)

In [38]:
question = "What are the main components of an LLM-powered autonomous agent system?"
questions = generate_queries_decomposition.invoke({"question":question})
questions = [x for x in questions if x != '']
len(questions)

3

In [39]:
 questions

['What are the key components of a large language model (LLM) used in autonomous decision-making?',
 'What role do natural language processing (NLP) and machine learning algorithms play in LLM-powered autonomous agents?',
 'How do LLMs integrate with other AI technologies, such as computer vision or reinforcement learning, to create a comprehensive autonomous system?']

## 1. Answer recursively
![image.png](attachment:dcb5b0ab-b047-43d4-99c0-b55889ed9af0.png)

In [31]:
# Prompt
template = """Here is the question you need to answer:

\n --- \n {question} \n --- \n

Here is any available background question + answer pairs:

\n --- \n {q_a_pairs} \n --- \n

Here is additional context relevant to the question: 

\n --- \n {context} \n --- \n

Use the above context and any background question + answer pairs to answer the question: \n {question}
"""

decomposition_prompt = ChatPromptTemplate.from_template(template)

In [32]:
from operator import itemgetter

# Retrieve

def format_qa_pair(question, answer):
    """Format Q and A pair"""
    
    formatted_string = ""
    formatted_string += f"Question: {question}\nAnswer: {answer}\n\n"
    return formatted_string.strip()
    
q_a_pairs = ""
for q in questions:
    
    rag_chain = (
    {"context": itemgetter("question") | retriever, 
     "question": itemgetter("question"),
     "q_a_pairs": itemgetter("q_a_pairs")} 
    | decomposition_prompt
    | llm
    | StrOutputParser())

    answer = rag_chain.invoke({"question":q,"q_a_pairs":q_a_pairs})
    q_a_pair = format_qa_pair(q,answer)
    q_a_pairs = q_a_pairs + "\n---\n"+  q_a_pair

In [33]:
answer

'Based on the provided context and background question + answer pairs, large language models (LLMs) can integrate with other AI technologies, such as computer vision or reinforcement learning, to create a comprehensive autonomous system in several ways:\n\n1. **Planning**: LLMs can be used for planning by breaking down complex tasks into smaller, manageable subgoals through techniques like subgoal and decomposition, reflection and refinement, task decomposition, and tree of thoughts. This allows the agent to plan ahead and make decisions autonomously.\n2. **Memory**: LLMs have limited context capacity, but mechanisms like vector stores and retrieval can provide access to a larger knowledge pool, enabling the agent to learn from past experiences and improve its decision-making.\n3. **Computer Vision**: Computer vision can be integrated with LLMs to enable the agent to understand and process visual inputs, such as images or videos. This can be done through techniques like object detectio

![image.png](attachment:12279d28-a763-4fa5-8b11-5e0758296612.png)

## 2. Answer individually
![image.png](attachment:92b96bf8-5f58-4591-adad-02eaf698db14.png)

In [40]:
from langchain import hub
from langchain_core.output_parsers import StrOutputParser

# RAG prompt
prompt_rag = hub.pull("rlm/rag-prompt")

def retrieve_and_rag(question,prompt_rag, sub_question_generator_chain):
    """RAG on each sub-question"""
    
    # Use our decomposition / 
    sub_questions = sub_question_generator_chain.invoke({"question":question})
    sub_questions = [x for x in sub_questions if x != '']
    
    # Initialize a list to hold RAG chain results
    rag_results = []
    
    for sub_question in sub_questions:
        
        # Retrieve documents for each sub-question
        retrieved_docs = retriever.get_relevant_documents(sub_question)
        
        # Use retrieved documents and sub-question in RAG chain
        answer = (prompt_rag | llm | StrOutputParser()).invoke({"context": retrieved_docs, 
                                                                "question": sub_question})
        rag_results.append(answer)
    
    return rag_results,sub_questions

answers, questions = retrieve_and_rag(question, prompt_rag, generate_queries_decomposition)

  retrieved_docs = retriever.get_relevant_documents(sub_question)


In [42]:
answers, questions

(['The key components of a large language model (LLM) used in autonomous decision-making are planning, reflection and refinement, and memory. Planning involves breaking down complex tasks into smaller subgoals, while reflection and refinement enable the agent to learn from mistakes and improve its performance over time. Memory plays a crucial role in storing historical information and providing access to a larger knowledge pool.',
  'NLP and machine learning algorithms play a crucial role in LLM-powered autonomous agents by enabling the agent to break down complex tasks into smaller subgoals, reflect on past actions, learn from mistakes, and refine its plans. NLP is used for task decomposition, planning, and self-reflection, while machine learning algorithms are employed for self-criticism, error detection, and adaptation to new situations. These algorithms help the agent to improve its performance over time through trial and error.',
  "LLMs integrate with other AI technologies throug

In [41]:
def format_qa_pairs(questions, answers):
    """Format Q and A pairs"""
    
    formatted_string = ""
    for i, (question, answer) in enumerate(zip(questions, answers), start=1):
        formatted_string += f"Question {i}: {question}\nAnswer {i}: {answer}\n\n"
    return formatted_string.strip()

context = format_qa_pairs(questions, answers)

In [43]:
# Prompt
template = """Here is a set of Q+A pairs:

{context}

Use these to synthesize an answer to the question: {question}
"""

prompt = ChatPromptTemplate.from_template(template)

In [44]:
final_rag_chain = (
    prompt
    | llm
    | StrOutputParser()
)

final_rag_chain.invoke({"context":context,"question":question})

"The main components of an LLM-powered autonomous agent system include planning, reflection and refinement, memory, natural language processing (NLP), machine learning algorithms, computer vision, and reinforcement learning. Planning involves breaking down complex tasks into smaller subgoals, while reflection and refinement enable the agent to learn from mistakes and improve its performance over time. Memory plays a crucial role in storing historical information and providing access to a larger knowledge pool.\n\nNLP is used for task decomposition, planning, and self-reflection, allowing the agent to break down complex tasks into smaller subgoals and reflect on past actions. Machine learning algorithms are employed for self-criticism, error detection, and adaptation to new situations, helping the agent to improve its performance over time through trial and error.\n\nComputer vision is integrated with LLMs to provide a comprehensive understanding of the environment, while reinforcement 

![image.png](attachment:1a622ef4-fb86-415b-abdc-63108f9f245a.png)

## Step-Back prompting

A simple prompting technique that enables LLMs to do abstractions to derive high-level concepts and first principles from instances containing specific details. Using the concepts and principles to guide reasoning, LLMs significantly improve their abilities in following a correct reasoning path towards the solution.

### Prompt

In [8]:
# Few Shot Examples
from langchain_core.prompts import ChatPromptTemplate, FewShotChatMessagePromptTemplate
examples = [
    {
        "input": "Could the members of The Police perform lawful arrests?",
        "output": "what can the members of The Police do?",
    },
    {
        "input": "Jan Sindel’s was born in what country?",
        "output": "what is Jan Sindel’s personal history?",
    },
]

# We now transform these to example messages
example_prompt = ChatPromptTemplate.from_messages(
    [
        ("human", "{input}"),
        ("ai", "{output}"),
    ]
)

few_shot_prompt = FewShotChatMessagePromptTemplate(
    example_prompt=example_prompt,
    examples=examples,
)

prompt = ChatPromptTemplate.from_messages(
    [
        (
            "system",
            """You are an expert at world knowledge. Your task is to step back and paraphrase a question to a more generic step-back question, which is easier to answer. Here are a few examples:""",
        ),
        # Few shot examples
        few_shot_prompt,
        # New question
        ("user", "{question}"),
    ]
)

In [12]:
from langchain_core.output_parsers import StrOutputParser

generate_queries_step_back = (
    prompt
    | llm
    | StrOutputParser()
)

In [13]:
question = "What is task decomposition for LLM agents?"
generate_queries_step_back.invoke({"question": question})

'how do large language models process complex tasks?'

In [14]:
# Response prompt 
response_prompt_template = """You are an expert of world knowledge. I am going to ask you a question. Your response should be comprehensive and not contradicted with the following context if they are relevant. Otherwise, ignore them if they are not relevant.

# {normal_context}
# {step_back_context}

# Original Question: {question}
# Answer:"""
response_prompt = ChatPromptTemplate.from_template(response_prompt_template)

In [18]:
chain = (
    {
        # Retrieve context using the normal question
        "normal_context": RunnableLambda(lambda x: x["question"]) | retriever,
        # Retrieve context using the step-back question
        "step_back_context": generate_queries_step_back | retriever,
        # Pass on the question
        "question": lambda x: x["question"],
    }
    | response_prompt
    | llm
    | StrOutputParser()
)

chain.invoke({"question": question})

'Task decomposition for LLM (Large Language Model) agents refers to the process of breaking down complex tasks into smaller, more manageable sub-tasks that can be solved by the agent. This allows the agent to focus on one task at a time and make progress towards completing the overall goal.\n\nIn the context of LLM agents, task decomposition is often achieved through the use of prompting techniques, such as chain-of-thought prompting or tree-of-thought prompting. These techniques involve providing the agent with a sequence of questions or prompts that guide it to break down the task into smaller sub-tasks and solve them one by one.\n\nFor example, in the article "LLM-powered Autonomous Agents" by Lilian Weng, task decomposition is discussed as a key component of LLM-based autonomous agents. The author suggests using techniques such as chain-of-thought prompting or tree-of-thought prompting to enable the agent to break down complex tasks into smaller sub-tasks and solve them one by one.

![image.png](attachment:b9eaa9d2-ab64-4990-a68d-5c7ba52c4599.png)

## HyDE

HyDE uses a Language Learning Model, to create a theoretical document when responding to a query, as opposed to using the query and its computed vector to directly seek in the vector database. Rather than seeking embedding similarity for questions or queries, it focuses on answer-to-answer embedding similarity.

![image.png](attachment:41d86e2b-29ca-4fa4-a1d1-d3c26f22792f.png)

### Prompt

In [19]:
# HyDE document genration
template = """Please write a scientific paper passage to answer the question
Question: {question}
Passage:"""
prompt_hyde = ChatPromptTemplate.from_template(template)

In [20]:
generate_docs_for_retrieval = (
    prompt_hyde
    | llm
    | StrOutputParser() 
)

In [21]:
question = "What is task decomposition for LLM agents?"
generate_docs_for_retrieval.invoke({"question":question})

"Here's a potential passage:\n\nTask Decomposition in Large Language Model (LLM) Agents: A Framework for Efficient and Effective Task Execution\n\nIn recent years, Large Language Model (LLM) agents have gained significant attention for their ability to perform complex tasks such as question answering, text generation, and dialogue management. However, the complexity of these tasks often leads to suboptimal performance due to the inherent limitations of LLMs in handling multi-step reasoning and task-specific knowledge. To address this challenge, researchers have proposed the concept of task decomposition for LLM agents.\n\nTask decomposition involves breaking down a complex task into smaller, more manageable sub-tasks that can be executed sequentially or in parallel by the LLM agent. This approach enables the agent to focus on one task at a time, reducing the cognitive load and improving its ability to reason about the task-specific knowledge required for each sub-task. By decomposing t

In [22]:
# Retrieve
retrieval_chain = generate_docs_for_retrieval | retriever 
retireved_docs = retrieval_chain.invoke({"question":question})
retireved_docs

[Document(id='c3d7430d-09d1-48c2-bb94-f42a4750b35b', metadata={'source': 'https://lilianweng.github.io/posts/2023-06-23-agent/'}, page_content='Fig. 1. Overview of a LLM-powered autonomous agent system.\nComponent One: Planning#\nA complicated task usually involves many steps. An agent needs to know what they are and plan ahead.\nTask Decomposition#\nChain of thought (CoT; Wei et al. 2022) has become a standard prompting technique for enhancing model performance on complex tasks. The model is instructed to “think step by step” to utilize more test-time computation to decompose hard tasks into smaller and simpler steps. CoT transforms big tasks into multiple manageable tasks and shed lights into an interpretation of the model’s thinking process.\nTree of Thoughts (Yao et al. 2023) extends CoT by exploring multiple reasoning possibilities at each step. It first decomposes the problem into multiple thought steps and generates multiple thoughts per step, creating a tree structure. The sear

In [23]:
template = """Answer the following question based on this context:

{context}

Question: {question}
"""
prompt = ChatPromptTemplate.from_template(template)

final_rag_chain = (
    prompt
    | llm
    | StrOutputParser()
)

In [24]:
final_rag_chain.invoke({"context":retireved_docs,"question":question})

'According to the text, task decomposition for LLM agents involves breaking down large tasks into smaller, manageable subgoals. This can be done in three ways:\n\n1. Using simple prompting, such as "Steps for XYZ. 1.", or "What are the subgoals for achieving XYZ?"\n2. Using task-specific instructions, such as "Write a story outline." for writing a novel\n3. With human inputs.\n\nTask decomposition is used to enable efficient handling of complex tasks and improve the quality of final results.'

![image.png](attachment:39c5f9d6-e6e7-43ce-94fe-1bfc0ba772ba.png)