In [None]:
from pathlib import Path
import os
from torch.cuda import is_available
import json

: 

In [None]:
from biokb.llm import AIRLLM
from biokb.agent import AIRAgent
from biokb.embedding import AIREmbedding, AIRPubmedSearch
from biokb.prompts import agent_prompt
from biokb.utils import get_file_names, create_documents_from_text_files, get_logger
from biokb.helpers import get_generation_config

from biokb.settings import MODEL_NAME, EMBEDDING_MODEL_NAME, DB_DIR, DATA_DIR

: 

Settings

In [3]:
logger = get_logger(debug=True)
logger.info("Starting the AIR Agent")

2024-10-13 12:05:01,506 - air-rag - INFO - Starting the AIR Agent


LLM Model

In [4]:
generation_config = get_generation_config()



In [5]:
llm = AIRLLM(
    model_name=MODEL_NAME,
    generation_config=generation_config,
    logger=logger
)

2024-10-13 12:05:01,659 - air-rag - INFO - Loading model: meta-llama/Meta-Llama-3.1-8B-Instruct
2024-10-13 12:05:01,662 - air-rag - INFO - Loading model in 4-bit mode
`low_cpu_mem_usage` was None, now set to True since model is quantized.


Loading checkpoint shards:   0%|          | 0/4 [00:00<?, ?it/s]

2024-10-13 12:05:07,585 - air-rag - INFO - Model loaded: meta-llama/Meta-Llama-3.1-8B-Instruct


In [6]:
def response(response: str):
    """
    A tool to respond to the user's queries.
    Args:
        response: The response string to respond to.
    """
    return response

def search_tool(query: str):
    """
    Use this tool to search for information in the database.
    Args:
        query: The query string to search for.
    """
    return "No information available"

In [7]:
embedding_llm = AIREmbedding(
    model_name=EMBEDDING_MODEL_NAME
)

In [8]:
if Path(DB_DIR).exists():
    docstore = AIRPubmedSearch.load(
        DB_DIR, 
        embedding_llm=embedding_llm,
    )
else:
    files = get_file_names(DATA_DIR)
    documents = create_documents_from_text_files(files)
    print(f"Number of documents: {len(documents)}")
    docstore = AIRPubmedSearch(
        documents=documents,
        embedding_llm=embedding_llm,
    )
    docstore.build(DB_DIR)

In [9]:
agent = AIRAgent(
    llm=llm,
    docstore=docstore,
    logger=logger,
    tools = [response, search_tool]
)

In [10]:
agent.invoke({"role":"user",
                   "content":"Hi"})

2024-10-13 12:05:10,243 - air-rag - DEBUG - Calling get_chat_response: [{'role': 'system', 'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."}, {'role': 'user', 'content': 'Hi'}]
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.
2024-10-13 12:05:12,605 - air-rag - INFO - Response: {"name": "response", "parameters": {"response": "Hello! How can I assist you today?"}}


({'role': 'assistant', 'content': 'Hello! How can I assist you today?'}, None)

In [11]:
agent.chat

[{'role': 'system',
  'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."},
 {'role': 'user', 'content': 'Hi'},
 {'role': 'assistant', 'content': 'Hello! How can I assist you today?'}]

In [12]:
message = agent.invoke({"role":"user",
                   "content":"what is mirna"})
message

2024-10-13 12:05:12,759 - air-rag - DEBUG - Calling get_chat_response: [{'role': 'system', 'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."}, {'role': 'user', 'content': 'Hi'}, {'role': 'assistant', 'content': 'Hello! How can I assist you today?'}, {'role': 'user', 'content': 'what is mirna'}]
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.
2024-10-13 12:05:20,657 - air-rag - INFO - Response: {"name": "response", "parameters": {"response": "MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression."}}


({'role': 'assistant',
  'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'},
 None)

In [13]:
agent.chat

[{'role': 'system',
  'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."},
 {'role': 'user', 'content': 'Hi'},
 {'role': 'assistant', 'content': 'Hello! How can I assist you today?'},
 {'role': 'user', 'content': 'what is mirna'},
 {'role': 'assistant',
  'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'}]

In [14]:
agent.trace

[{'role': 'user', 'content': 'Hi'},
 {'role': 'assistant', 'content': 'Hello! How can I assist you today?'},
 {'role': 'user', 'content': 'what is mirna'},
 {'role': 'assistant',
  'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'}]

In [15]:
message = agent.invoke({"role":"user",
                   "content":"ok"})

2024-10-13 12:05:20,864 - air-rag - DEBUG - Calling get_chat_response: [{'role': 'system', 'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."}, {'role': 'user', 'content': 'Hi'}, {'role': 'assistant', 'content': 'Hello! How can I assist you today?'}, {'role': 'user', 'content': 'what is mirna'}, {'role': 'assistant', 'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'}, {'role': 'user', 'content': 'ok'}]
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.
2024-10-13 12:05:23,950 - air-rag - INFO - Response: {"name": 

In [16]:
message

({'role': 'assistant',
  'content': 'MicroRNA is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression.'},
 None)

In [17]:
agent.chat

[{'role': 'system',
  'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."},
 {'role': 'user', 'content': 'Hi'},
 {'role': 'assistant', 'content': 'Hello! How can I assist you today?'},
 {'role': 'user', 'content': 'what is mirna'},
 {'role': 'assistant',
  'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'},
 {'role': 'user', 'content': 'ok'},
 {'role': 'assistant',
  'content': 'MicroRNA is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression.'}]

In [18]:
agent.trace

[{'role': 'user', 'content': 'Hi'},
 {'role': 'assistant', 'content': 'Hello! How can I assist you today?'},
 {'role': 'user', 'content': 'what is mirna'},
 {'role': 'assistant',
  'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'},
 {'role': 'user', 'content': 'ok'},
 {'role': 'assistant',
  'content': 'MicroRNA is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression.'}]

In [19]:
message = agent.invoke({"role":"user",
                   "content":"haha"})
message

2024-10-13 12:05:24,217 - air-rag - DEBUG - Calling get_chat_response: [{'role': 'system', 'content': "You are a chat agent. Use 'response' tool to generate responses. Use 'search' tool to search before you asnwer any scientific questions. The query should have some detail not just phrases or keywords. Do not give user any scientific answer without using search tool."}, {'role': 'user', 'content': 'Hi'}, {'role': 'assistant', 'content': 'Hello! How can I assist you today?'}, {'role': 'user', 'content': 'what is mirna'}, {'role': 'assistant', 'content': 'MicroRNA (miRNA) is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional regulation of gene expression. It plays a crucial role in various biological processes, including development, cell differentiation, and disease progression.'}, {'role': 'user', 'content': 'ok'}, {'role': 'assistant', 'content': 'MicroRNA is a small non-coding RNA molecule that functions in RNA silencing and post-transcriptional 

({'role': 'assistant', 'content': 'It looks like we didn'}, None)