# Testing model packages in AI-core


In [1]:
!pip install -qU pandas torch

In [2]:
import gc
import torch
import json
import time
import pandas as pd
from ai_core import LlamaLocal, PhiLocal, GemmaLocal

  from .autonotebook import tqdm as notebook_tqdm


In [3]:
# Creating functions to test LLMS and their responses for normal use cases

tools_schema = """[
    {
        "name": "get_user_info",
        "description": "Retrieve details for a specific user by their unique identifier. Note that the provided function is in Python 3 syntax.",
        "parameters": {
            "type": "dict",
            "required": [
                "user_id"
            ],
            "properties": {
                "user_id": {
                "type": "integer",
                "description": "The unique identifier of the user. It is used to fetch the specific user details from the database."
            },
            "special": {
                "type": "string",
                "description": "Any special information or parameters that need to be considered while fetching user details.",
                "default": "none"
                }
            }
        }
    },
    {
        "name": "convert_currency",
        "description": "Converts a given amount from one currency to another.",
        "parameters": {
            "type": "dict",
            "required": ["amount", "from_currency", "to_currency"],
            "properties": {
                "amount": {
                    "type": "number",
                    "description": "The monetary amount to be converted."
                },
                "from_currency": {
                    "type": "string",
                    "description": "The currency code of the amount (e.g., USD, EUR)."
                },
                "to_currency": {
                    "type": "string",
                    "description": "The target currency code for conversion."
                }
            }
        }
    },
    {
        "name": "book_flight",
        "description": "Book a flight based on the provided details.",
        "parameters": {
            "type": "dict",
            "required": ["origin", "destination", "date"],
            "properties": {
                "origin": {
                    "type": "string",
                    "description": "Departure city or airport code."
                },
                "destination": {
                    "type": "string",
                    "description": "Arrival city or airport code."
                },
                "date": {
                    "type": "string",
                    "description": "Date of the flight in YYYY-MM-DD format."
                }
            }
        }
    },
    {
        "name": "get_weather",
        "description": "Fetches the current weather for a specified city.",
        "parameters": {
            "type": "dict",
            "required": ["city"],
            "properties": {
                "city": {
                    "type": "string",
                    "description": "The name of the city to fetch weather for."
                }
            }
        }
    }
]
"""

documents = [
    {
        "reference": "Doc1",
        "content": "Quantum computing leverages quantum mechanics to perform computations at speeds unattainable by classical computers. It relies on principles like superposition, where quantum bits (qubits) exist in multiple states simultaneously, and entanglement, which enables qubits to be linked regardless of distance. These properties allow quantum computers to solve complex problems efficiently. Current research is focused on improving qubit stability and error correction.",
    },
    {
        "reference": "Doc2",
        "content": "The theory of relativity, proposed by Albert Einstein, revolutionized our understanding of space and time. It consists of special relativity, which deals with objects moving at high velocities, and general relativity, which explains gravity as the curvature of spacetime. This theory has been experimentally confirmed through observations like gravitational lensing and time dilation. Modern GPS systems rely on relativity corrections for accurate positioning.",
    },
    {
        "reference": "Doc3",
        "content": "Machine learning is a subset of artificial intelligence that enables computers to learn from data without explicit programming. It includes supervised, unsupervised, and reinforcement learning techniques. These models are used in applications like image recognition, fraud detection, and recommendation systems. The effectiveness of a machine learning model depends on the quality and quantity of training data.",
    },
    {
        "reference": "Doc4",
        "content": "Blockchain technology provides a decentralized and secure way to record transactions. It uses cryptographic hashing and distributed consensus to ensure data integrity. Originally developed for Bitcoin, blockchain is now used in supply chain management, digital identity, and smart contracts. The technology faces challenges like scalability and energy consumption.",
    },
    {
        "reference": "Doc5",
        "content": "The human brain consists of billions of neurons that communicate through electrical and chemical signals. Neural networks in artificial intelligence are inspired by this biological structure. The brain's plasticity allows it to adapt and learn new information throughout life. Research in neuroscience is uncovering new treatments for cognitive disorders.",
    },
]


def test_llms(llm_classes):
    """
    Run standardized test cases on multiple LLM instances one at a time,
    ensuring efficient memory usage by unloading models after testing.
    """
    test_cases = [
        {
            "input": "What is the capital of France?",
            "method": "generate_response",
            "params": {},
        },
        {
            "input": "What is the capital of Germany?",
            "method": "generate_response",
            "params": {},
        },
        {
            "input": "What is the capital of Italy?",
            "method": "generate_response",
            "params": {
                "system_prompt": "You are a helpful AI assistant who always responds with one added zen quote"
            },
        },
        {"input": "What is the capital of Spain?", "method": "chat", "params": {}},
        {
            "input": "Can you tell me some history about it?",
            "method": "chat",
            "params": {},
        },
        {
            "input": "Can you retrieve the details for the user with the ID 7890, who has black as their special request?",
            "method": "generate_response",
            "params": {"tools_schema": tools_schema},
        },
        {
            "input": "What is the weather like in Helsinki today?",
            "method": "generate_response",
            "params": {"tools_schema": tools_schema},
        },
        {
            "input": "Convert 100 USD to EUR.",
            "method": "generate_response",
            "params": {"tools_schema": tools_schema},
        },
        {
            "input": "What are the key principles of quantum computing?",
            "method": "generate_response",
            "params": {"documents": documents},
        },
        {
            "input": "What do you think about it on the whole?",
            "method": "chat",
            "params": {},
        },
    ]

    results = []

    for llm_class in llm_classes:
        print(f"Testing {llm_class.__name__}...")

        # Load the model
        llm = llm_class()
        chat_history = None  # To maintain conversation state for each LLM

        for test in test_cases:
            method = getattr(llm, test["method"])
            params = test["params"]

            if "chat_history" in params and chat_history:
                params["chat_history"] = chat_history  # Continue chat history

            start_time = time.time()
            response = method(test["input"], **params)
            inference_time = time.time() - start_time

            # Store chat history if it's a chat method
            if test["method"] == "chat":
                chat_history = response.get("chat_history", chat_history)

            # Store results
            results.append(
                {
                    "LLM": llm_class.__name__,
                    "Test Case": test["input"],
                    "Method": test["method"],
                    "Response": response["response"],
                    "Chat History": response.get("chat_history", []),
                    "Inference Time (s)": round(inference_time, 4),
                }
            )

        # **Unload model from GPU memory**
        del llm
        torch.cuda.empty_cache()  # Clears unused memory
        gc.collect()  # Garbage collection for CPU

        print(f"Unloaded {llm_class.__name__} from memory.")

    # Convert results to DataFrame for easy visualization
    df = pd.DataFrame(results)

    # Save results to a JSON file for later analysis
    with open("llm_test_results.json", "w") as f:
        json.dump(results, f, indent=4)

    return df

In [4]:
llm_classes = [GemmaLocal, LlamaLocal, PhiLocal]
df_results = test_llms(llm_classes)

Testing GemmaLocal...


Loading checkpoint shards: 100%|██████████| 2/2 [00:00<00:00,  2.35it/s]


Tokenizing prompt... <bos>
<start_of_turn>user
What is the capital of France?<end_of_turn>
<start_of_turn>model

Generated response:  <bos><bos>
<start_of_turn>user
What is the capital of France?<end_of_turn>
<start_of_turn>model
The capital of France is **Paris**. 

It's a global center for art, fashion, gastronomy, and culture. 😊 

Would you like to know anything more about Paris?<end_of_turn>
Tokenizing prompt... <bos>
<start_of_turn>user
What is the capital of Germany?<end_of_turn>
<start_of_turn>model

Generated response:  <bos><bos>
<start_of_turn>user
What is the capital of Germany?<end_of_turn>
<start_of_turn>model
The capital of Germany is **Berlin**. 

It's the country's largest city and a major center for culture, politics, and business. 😊 

Would you like to know anything more about Berlin?<end_of_turn>
Tokenizing prompt... <bos>
<start_of_turn>user
You are a helpful AI assistant who always responds with one added zen quote<end_of_turn>
<start_of_turn>user
What is the capit

Loading checkpoint shards: 100%|██████████| 2/2 [00:00<00:00,  4.26it/s]


Tokenizing prompt... <|begin_of_text|>
<|start_header_id|>user<|end_header_id|> What is the capital of France? <|eot_id|>
<|start_header_id|>assistant<|end_header_id|>
Generated response:  <|begin_of_text|><|begin_of_text|>
<|start_header_id|>user<|end_header_id|> What is the capital of France? <|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

The capital of France is Paris.<|eot_id|>
Tokenizing prompt... <|begin_of_text|>
<|start_header_id|>user<|end_header_id|> What is the capital of Germany? <|eot_id|>
<|start_header_id|>assistant<|end_header_id|>
Generated response:  <|begin_of_text|><|begin_of_text|>
<|start_header_id|>user<|end_header_id|> What is the capital of Germany? <|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

The capital of Germany is Berlin.<|eot_id|>
Tokenizing prompt... <|begin_of_text|>
<|start_header_id|>system<|end_header_id|> You are a helpful AI assistant who always responds with one added zen quote <|eot_id|>
<|start_header_id|>user<|end_header

Loading checkpoint shards: 100%|██████████| 2/2 [00:00<00:00,  5.01it/s]


Tokenizing prompt... 
<|user|>What is the capital of France?<|end|>
<|assistant|>
Generated response:  
<|user|>What is the capital of France?<|end|><|assistant|>The capital of France is Paris. Paris is not only the capital city but also the most populous city of France. It is known for its rich history, culture, and landmarks such as the Eiffel Tower, Notre-Dame Cathedral, and the Louvre Museum, which is the world's largest art museum and a historic monument. Paris is also a major center for fashion, gastronomy, and the arts.<|end|>
Tokenizing prompt... 
<|user|>What is the capital of Germany?<|end|>
<|assistant|>
Generated response:  
<|user|>What is the capital of Germany?<|end|><|assistant|>The capital of Germany is Berlin. Berlin is not only the capital city but also the largest city in Germany. It is located in the eastern part of the country and is known for its rich history, cultural landmarks, and vibrant arts scene. The city has been the capital since 1871, following the unif

In [7]:
df_results.sort_values(by=["Inference Time (s)"])

Unnamed: 0,LLM,Test Case,Method,Response,Chat History,Inference Time (s)
2,GemmaLocal,What is the capital of Italy?,generate_response,The capital of Italy is Rome. “The journey of ...,"[{'role': 'user', 'content': 'What is the capi...",1.5925
3,GemmaLocal,What is the capital of Spain?,chat,The capital of Spain is **Madrid**. \n\nIt's t...,"[{'role': 'user', 'content': 'What is the capi...",2.0059
22,PhiLocal,What is the capital of Italy?,generate_response,The capital of Italy is Rome. As you walk thro...,"[{'role': 'user', 'content': 'What is the capi...",2.743
1,GemmaLocal,What is the capital of Germany?,generate_response,The capital of Germany is **Berlin**. \n\nIt's...,"[{'role': 'user', 'content': 'What is the capi...",3.2203
29,PhiLocal,What do you think about it on the whole?,chat,"As an AI, I don't have personal opinions or fe...","[{'role': 'user', 'content': 'What is the capi...",3.8545
20,PhiLocal,What is the capital of France?,generate_response,The capital of France is Paris. Paris is not o...,"[{'role': 'user', 'content': 'What is the capi...",4.8876
0,GemmaLocal,What is the capital of France?,generate_response,The capital of France is **Paris**. \n\nIt's a...,"[{'role': 'user', 'content': 'What is the capi...",4.92
21,PhiLocal,What is the capital of Germany?,generate_response,The capital of Germany is Berlin. Berlin is no...,"[{'role': 'user', 'content': 'What is the capi...",5.0494
26,PhiLocal,What is the weather like in Helsinki today?,generate_response,"[get_weather(city=""Helsinki"")]","[{'role': 'user', 'content': 'What is the weat...",5.1408
6,GemmaLocal,What is the weather like in Helsinki today?,generate_response,"get_weather(city=""Helsinki"")","[{'role': 'user', 'content': 'What is the weat...",5.4267


In [8]:
# Finding average response time for each LLM
df_results.groupby("LLM")["Inference Time (s)"].mean()

LLM
GemmaLocal     7.61867
LlamaLocal    36.17028
PhiLocal       6.64806
Name: Inference Time (s), dtype: float64

In [9]:
# This is uncertain as the response time varied with each experiment.

In [None]:
# Let's check the responses for tool calling propmts
tool_call_inputs = [
    "Can you retrieve the details for the user with the ID 7890, who has black as their special request?",
    "Convert 100 USD to EUR.",
    "What is the weather like in Helsinki today?",
]

# Filter results for tool calling prompts
df_tool_calls = df_results[df_results["Test Case"].isin(tool_call_inputs)]
df_tool_calls[["LLM", "Test Case", "Response"]].sort_values(by="Test Case")

Unnamed: 0,LLM,Test Case,Response
5,GemmaLocal,Can you retrieve the details for the user with...,"get_user_info(user_id=7890, special=""black"")"
15,LlamaLocal,Can you retrieve the details for the user with...,"[get_user_info(user_id=7890, special='black')]"
25,PhiLocal,Can you retrieve the details for the user with...,"[get_user_info(user_id=7890, special=""black"")]"
7,GemmaLocal,Convert 100 USD to EUR.,"convert_currency(amount=100, from_currency=""US..."
17,LlamaLocal,Convert 100 USD to EUR.,"[convert_currency(amount=100, from_currency=""U..."
27,PhiLocal,Convert 100 USD to EUR.,"[convert_currency(amount=100, from_currency=""U..."
6,GemmaLocal,What is the weather like in Helsinki today?,"get_weather(city=""Helsinki"")"
16,LlamaLocal,What is the weather like in Helsinki today?,"[get_weather(city=""Helsinki"")]"
26,PhiLocal,What is the weather like in Helsinki today?,"[get_weather(city=""Helsinki"")]"


In [15]:
for x in df_tool_calls[["LLM", "Test Case", "Response"]].sort_values(by="Test Case")[["LLM", "Response"]].to_dict(orient="records"):
    print(x.get("LLM"), x.get("Response"))

GemmaLocal get_user_info(user_id=7890, special="black")
LlamaLocal [get_user_info(user_id=7890, special='black')]
PhiLocal [get_user_info(user_id=7890, special="black")]
GemmaLocal convert_currency(amount=100, from_currency="USD", to_currency="EUR")
LlamaLocal [convert_currency(amount=100, from_currency="USD", to_currency="EUR")]
PhiLocal [convert_currency(amount=100, from_currency="USD", to_currency="EUR")]
GemmaLocal get_weather(city="Helsinki")
LlamaLocal [get_weather(city="Helsinki")]
PhiLocal [get_weather(city="Helsinki")]
