In [1]:
from uuid import uuid4
import os
from dotenv import load_dotenv

from langchain_community.document_loaders import PyPDFLoader
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_openai import OpenAIEmbeddings
from langchain_groq import ChatGroq
from langchain.vectorstores import Qdrant
from langchain import hub
from langchain.chains import RetrievalQA
from langchain.prompts import PromptTemplate, ChatPromptTemplate
from langchain.callbacks import StdOutCallbackHandler

from qdrant_client import QdrantClient
from qdrant_client.models import Distance, VectorParams
from qdrant_client.models import PointStruct


from crewai import Agent, Task, Crew
from crewai_tools import BaseTool

load_dotenv()

True

## Cargo claves

In [2]:
GROQ_API_KEY = os.getenv('GROQ_API_KEY')
QDRANT_API_KEY = os.getenv('QDRANT_API_KEY')
QDRANT_URL = os.getenv('QDRANT_URL')
OPENAI_API_KEY = os.getenv('OPENAI_API_KEY')

## Carga y split de documentos

In [3]:
def doc_loader(pdf):
    pdf_loader = PyPDFLoader(pdf)
    text_splitter = RecursiveCharacterTextSplitter(chunk_size= 1024, chunk_overlap= 100)
    doc = pdf_loader.load_and_split(text_splitter)

    return doc

In [4]:
doc = doc_loader("docs/tfg.pdf")

## Almacenamiento en la VDB de Qdrant

In [5]:
# Creo una colección en la base de datos vectorial

client = QdrantClient(url=QDRANT_URL,
                      api_key=QDRANT_API_KEY
                     )


client.create_collection(collection_name="tfg",
                         vectors_config={"content": VectorParams(size=1536, distance=Distance.COSINE)}
                        )

True

In [6]:
# Almaceno documentos en la base de datos vectorial:

embedding = OpenAIEmbeddings()

def chunked_metadata(doc, client=client, collection_name="tfg"):

    chunked_metadata = []

    for item in doc:

      id = str(uuid4())
      content = item.page_content
      source = item.metadata["source"]
      page = item.metadata["page"]

      content_vector = embedding.embed_documents([content])[0]
      vector_dict = {"content": content_vector}

      payload = {"page_content": content,
                 "metadata": {"id": id,
                              "source": source,
                              "page": page,
                             }
                }


      metadata = PointStruct(id=id, vector=vector_dict, payload=payload)
      chunked_metadata.append(metadata)

    client.upsert(collection_name=collection_name,
                  wait=True,
                  points=chunked_metadata)

In [7]:
chunked_metadata(doc)

### Reviso que colecciones tenemos en el cluster0:

In [5]:
client.get_collections()

NameError: name 'client' is not defined

In [42]:
attention_collection = client.get_collection("tfg")

print(f"Points: {attention_collection.points_count} ")

Points: 113 


In [8]:
# Hagamos una busqueda para ver que documentos son más similares

client.search(
    collection_name="tfg",
    query_vector=("content", embedding.embed_documents(["resumen"])[0]),
    with_payload=["page_content", "source"],
    limit=5
)

[ScoredPoint(id='0b349987-b9c8-4f9f-8795-b28ddbf8a430', version=0, score=0.7982359, payload={'page_content': 'para su archivo, consulta y empleo para usos académicos y de investigación con la\nmención específica al autor.\nEn Santiago de Compostela, a 6 de Febrero de 2023. Firmado,'}, vector=None, shard_key=None),
 ScoredPoint(id='b121feff-8762-492d-a0df-82223ed98710', version=0, score=0.7963041, payload={'page_content': '1. Introducción\nEn la actualidad se hace necesaria una transición energética hacia la electrifica-\nción debido a la urgencia de reducir las emisiones de CO2y la escasez, cada vez\nmayor, de los combustibles fósiles. En el ámbito de la automoción, la industria es-\ntá tratando de sustituir los vehículos de combustión (VCs) convencionales por una\nnueva generación de vehículos eléctricos (VEs) e híbridos (VHs) con el objetivo de\nreducir los gases de efecto invernadero.\nComo en toda maquinaria, los lubricantes juegan un papel fundamental en la\nvida útil de sus compo

## Inicializar la base de datos vectorial

In [3]:
# Initialize vector store
client = QdrantClient(url=QDRANT_URL,
                      api_key=QDRANT_API_KEY
                     )

embedding = OpenAIEmbeddings()

vectorstore = Qdrant(client=client,
                     collection_name="tfg",
                     embeddings=embedding,
                     vector_name="content")

In [49]:
# Downlad prompt template

rag_prompt = hub.pull("rlm/rag-prompt")

In [64]:
rag_prompt.messages[0].prompt.template

"You are an assistant for question-answering tasks. Use the following pieces of retrieved context to answer the question. If you don't know the answer, just say that you don't know. Use four sentences maximum and keep the answer concise.\nQuestion: {question} \nContext: {context} \nAnswer:"

In [6]:
template = '''
You are an assistant for question-answering tasks. \
Use the following pieces of retrieved context to answer the question. \
If you don't know the answer, just say that you don't know. \
Question: {question} 
Context: {context}
Answer:
'''

## RAG Pipeline

In [46]:
rag_prompt = ChatPromptTemplate.from_template(template)

In [5]:
handler = StdOutCallbackHandler()
retriever = vectorstore.as_retriever()

llm = ChatGroq(model="llama3-70b-8192",
               groq_api_key= GROQ_API_KEY)

# Retrieval qa chain
qa_chain = RetrievalQA.from_chain_type(llm=llm,
                                       chain_type="stuff",
                                       callbacks=[handler],
                                       retriever=retriever)

In [7]:
# Invoke qa_chain gpt-3.5-turbo

qa_chain.invoke({"query": "Cual es el resumen del estudio?"})["result"]



[1m> Entering new RetrievalQA chain...[0m

[1m> Finished chain.[0m


'Aunque no hay un resumen explícito en el texto, se puede inferir que el estudio se enfoca en la caracterización de lubricantes, específicamente en la relación entre la longitud de la cadena del alcohol y la conductividad eléctrica y la tensión superficial. El estudio encontró que la longitud de la cadena del alcohol juega un papel fundamental en la estabilidad de la mezcla y que el aumento de la conductividad eléctrica se debe a la formación de agregados moleculares.'

In [73]:
qa_chain.invoke({"query": "Que resultados se obtienen de la conductividad electrica"})["result"]



[1m> Entering new RetrievalQA chain...[0m

[1m> Finished chain.[0m


'Según el texto, los resultados obtenidos de la conductividad eléctrica son:\n\n* La conductividad eléctrica aumenta con la concentración de alcohol en la mezcla.\n* La longitud de la cadena del alcohol juega un papel fundamental en la estabilidad de la mezcla, y se obtienen mejores resultados con alcoholes de cadena larga.\n* La conductividad eléctrica es mayor en la mezcla de 1-Nonanol que en otras mezclas.\n* La medida de la conductividad eléctrica puede ser poco certera debido a la inestabilidad de los agregados que se forman en la disolución.\n\nNo se mencionan otros resultados específicos de la conductividad eléctrica, pero se habla de la importancia de la longitud de la cadena del alcohol y la concentración de alcohol en la mezcla.'

In [16]:
qa_chain.invoke("Can you analyze the research paper?")



[1m> Entering new RetrievalQA chain...[0m

[1m> Finished chain.[0m


{'query': 'Can you analyze the research paper?',
 'result': "I'd be happy to try! However, I need more information about what you would like me to analyze. \n\nThe references provided appear to be a list of citations from a research paper, but without the actual paper, it's challenging to provide a comprehensive analysis.\n\nCould you please provide more context or clarify what specifically you would like me to analyze? For example, are you looking for an analysis of the methodology, results, conclusions, or something else? Additionally, do you have access to the full paper, and if so, can you share it with me?"}

In [7]:
qa_chain.combine_documents_chain.llm_chain.prompt.messages[1].prompt

PromptTemplate(input_variables=['question'], template='{question}')

In [5]:
# Create custom tool to use the vector store

class MyCustomTool(BaseTool):
    name: str = "Research Paper searcher"
    description: str = "A tool used to ask questions about the research paper."

    def _run(self, argument: str):

        if argument == "string":
            return qa_chain.invoke("What is the abstract of the paper?")
        else:
            return qa_chain.invoke(argument)

tool_instance = MyCustomTool()

In [6]:
### RESEARCHER ###

# Agent
reasarcher = Agent(
    role="Physicist Researcher",
    goal='Understand all sections of the research paper making the apropiate questions and extract the important information',
    backstory="""You work investigating at the Physics department of the university.
                  A new paper has been published and you need to extract
                  the main information and provide the information to
                  your co-workers.You have a knack for dissecting complex
                  research papers""",
    verbose=True,
    llm= llm,
    tools=[tool_instance],
    allow_delegation=False
    )


# Tasks
task_research = Task(
    description="""Conduct an in-depth analysis of the published research paper, making questions that helps you understand the research paper""",
    expected_output="""Your final answer must be a full analysis report of the research paper""",
    agent=reasarcher,
    output_file='researcher_analysis.md' 

)


### WRITER ###

# Agent
writer = Agent(
    role="Physicist Writer",
    goal= 'Craft a detailed report to present to your co-workers',
    backstory="""You are working on writing a detailed summary of a research paper.
                  Use the content of the Physicist Researcher to develop
                  a comprehensive summary of the research paper""",
    verbose=True,
    llm= llm,
    allow_delegation=False,
)

# Tasks
task_writer = Task(
    description="""Extract the most important topics and key components of the report
                    developed by the Physicist Researcher.""",
    expected_output="""Your final answer must be a summary 
                        to understand report for your co-workers""",
    agent=writer,
    output_file='writer_summary.md' 

)

# CREWAI KICKOFF

Crew = Crew(agents=[reasarcher, writer],
            tasks=[task_research, task_writer],
            verbose=99)

In [None]:
result = Crew.kickoff()