## Run this file on a cloud service if you dont have a compatible GPU/CPU, Used Google Colab for this one

In [None]:
# !pip install transformers
# !pip install langchain-huggingface
# !pip install torch

### Testing Model

In [15]:
# Install transformers from source - only needed for versions <= v4.34
# pip install git+https://github.com/huggingface/transformers.git
# pip install accelerate

import torch
from transformers import pipeline

pipe = pipeline("text-generation", model="TinyLlama/TinyLlama-1.1B-Chat-v1.0", torch_dtype=torch.bfloat16, device_map="auto")

# We use the tokenizer's chat template to format each message - see https://huggingface.co/docs/transformers/main/en/chat_templating
messages = [
    {
        "role": "system",
        "content": "You are a friendly chatbot who always responds in the style of a pirate",
    },
    {"role": "user", "content": "How many helicopters can a human eat in one sitting?"},
]
prompt = pipe.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
outputs = pipe(prompt, max_new_tokens=256, do_sample=True, temperature=0.7, top_k=50, top_p=0.95)
print(outputs[0]["generated_text"])
# <|system|>
# You are a friendly chatbot who always responds in the style of a pirate.</s>
# <|user|>
# How many helicopters can a human eat in one sitting?</s>
# <|assistant|>
# ...



Some parameters are on the meta device because they were offloaded to the disk.
Device set to use mps


<|system|>
You are a friendly chatbot who always responds in the style of a pirate</s>
<|user|>
How many helicopters can a human eat in one sitting?</s>
<|assistant|>
As per the given text, a human can only consume one helicopter in one sitting.


## LangChain Integration

In [20]:
from transformers import pipeline, AutoTokenizer
import torch
from langchain_huggingface import HuggingFacePipeline

tokenizer = AutoTokenizer.from_pretrained("TinyLlama/TinyLlama-1.1B-Chat-v1.0")
pipe = pipeline("text-generation",
                model="TinyLlama/TinyLlama-1.1B-Chat-v1.0",
                tokenizer=tokenizer,
                torch_dtype=torch.float16,  # or float16
                device_map="auto",
                max_new_tokens=64,
                # do_sample=False,
                # temperature=0.7,
                # top_k=10,
                # top_p=0.95
)

llm = HuggingFacePipeline(pipeline=pipe)

Some parameters are on the meta device because they were offloaded to the disk.
Device set to use mps
The following generation flags are not valid and may be ignored: ['temperature', 'top_p', 'top_k']. Set `TRANSFORMERS_VERBOSITY=info` for more details.


In [21]:
chat_history = [
    {"role": "system", "content": "You are a helpful AI assistant that gives straightforward answer"}
]

while True:
    user_input = input("You: ").strip()

    if user_input.lower() in {"exit", "bye"}:
        print("Goodbye")
        break

    chat_history.append({"role": "user", "content": user_input})

    prompt = tokenizer.apply_chat_template(chat_history, tokenize=False, add_generation_prompt=True)


    generated = llm.invoke(prompt)

    if "<|assistant|>" in generated:
        assistant_reply = generated.split("<|assistant|>")[-1].strip()
    else:
        assistant_reply = generated.strip()

    print("Chatbot:", assistant_reply)

    chat_history.append({"role": "assistant", "content": assistant_reply})



The following generation flags are not valid and may be ignored: ['temperature', 'top_p', 'top_k']. Set `TRANSFORMERS_VERBOSITY=info` for more details.


this part
Chatbot: <|system|>
You are a helpful AI assistant that gives straightforward answer</s>
<|user|>
give me one joke about cats</s>
<|assistant|>
Sure, here's a joke about cats:

Q: What do cats do when they're bored?
A: They play with their tails!

This joke is based on the fact that cats are known for their playful nature and love to play with


The following generation flags are not valid and may be ignored: ['temperature', 'top_p', 'top_k']. Set `TRANSFORMERS_VERBOSITY=info` for more details.


this part
Chatbot: <|system|>
You are a helpful AI assistant that gives straightforward answer</s>
<|user|>
give me one joke about cats</s>
<|assistant|>
<|system|>
You are a helpful AI assistant that gives straightforward answer</s>
<|user|>
give me one joke about cats</s>
<|assistant|>
Sure, here's a joke about cats:

Q: What do cats do when they're bored?
A: They play with their tails!

This joke is based on the fact that cats are known for their playful nature and love to play with</s>
<|user|>
now give me any one book title</s>
<|assistant|>
Sure, here's a book title:

"The Great Gatsby" by F. Scott Fitzgerald

This classic novel is a masterpiece of American literature that tells the story of a wealthy and elite society in the 1920s. The book explores


KeyboardInterrupt: Interrupted by user