In [1]:
from transformers import AutoTokenizer, AutoModelForCausalLM,BitsAndBytesConfig
import torch

torch.cuda.is_available()

True

In [2]:
import langchain

In [3]:
quantization_config = BitsAndBytesConfig(load_in_4bit=True)
tokenizer = AutoTokenizer.from_pretrained("google/gemma-2-9b-it")
model = AutoModelForCausalLM.from_pretrained(
    "google/gemma-2-9b-it",
    device_map="cuda",
    quantization_config=quantization_config
)

Loading checkpoint shards:   0%|          | 0/4 [00:00<?, ?it/s]

In [4]:
class model_chat:
    def __init__(self,model,tokenizer):
        self.model = model
        self.tokenizer = tokenizer
        self.chat_history = []
        self.special_tokens = ['<bos>','<eos>','<pad>','<unk>','<start_of_turn>','<end_of_turn>']

    def send_message(self,message,max_new_tokens=100) -> str:
        self.chat_history.append({"role":"user","content":message})
        input_text = tokenizer.apply_chat_template(self.chat_history,tokenize=False,add_generation_prompt=True)
        print(input_text)
        input_ids = tokenizer(input_text, add_special_tokens=False,return_tensors="pt").to("cuda")
        outputs = model.generate(**input_ids,max_new_tokens=max_new_tokens)
        text = tokenizer.decode(outputs[0])

        model_message = text.replace(input_text,"")
        for special_token in self.special_tokens:
            model_message = model_message.replace(special_token,"")
        
        self.chat_history.append({"role":"model","content":model_message})
        return model_message

In [5]:
chat = model_chat(model,tokenizer)
print(chat.send_message("My name is mark bark, nice to meet you!"))

<bos><start_of_turn>user
My name is mark bark, nice to meet you!<end_of_turn>
<start_of_turn>model





Nice to meet you too, Mark Bark! It's a pleasure to make your acquaintance.

Is there anything I can help you with today?



In [6]:
print(chat.send_message("Can you write a hello world program in rust?"))

<bos><start_of_turn>user
My name is mark bark, nice to meet you!<end_of_turn>
<start_of_turn>model
Nice to meet you too, Mark Bark! It's a pleasure to make your acquaintance.

Is there anything I can help you with today?<end_of_turn>
<start_of_turn>user
Can you write a hello world program in rust?<end_of_turn>
<start_of_turn>model

```rust
fn main() {
    println!("Hello, world!");
}
```

**Explanation:**

* `fn main()`: This line defines the main function, which is the entry point of every Rust program.
* `println!("Hello, world!");`: This line uses the `println!` macro to print the string "Hello, world!" to the console.

**To run this program:**

1. Save the code as a file named `hello_world


In [7]:
print(chat.send_message("Can you remind me my name?"))

<bos><start_of_turn>user
My name is mark bark, nice to meet you!<end_of_turn>
<start_of_turn>model
Nice to meet you too, Mark Bark! It's a pleasure to make your acquaintance.

Is there anything I can help you with today?<end_of_turn>
<start_of_turn>user
Can you write a hello world program in rust?<end_of_turn>
<start_of_turn>model
```rust
fn main() {
    println!("Hello, world!");
}
```

**Explanation:**

* `fn main()`: This line defines the main function, which is the entry point of every Rust program.
* `println!("Hello, world!");`: This line uses the `println!` macro to print the string "Hello, world!" to the console.

**To run this program:**

1. Save the code as a file named `hello_world<end_of_turn>
<start_of_turn>user
Can you remind me my name?<end_of_turn>
<start_of_turn>model

You told me your name is Mark Bark.  

Is there anything else I can help you with?



In [23]:
chat_history = [{"role":"user","content":"My name is mark bark, nice to meet you!"},{"role":"assistant","content":"Hello mark bark,"}]
input_text = tokenizer.apply_chat_template(chat_history[:-1],tokenize=False,add_generation_prompt=False) + f"<start_of_turn>model\n{chat_history[-1]['content']}"
input_ids = tokenizer(input_text, add_special_tokens=False,return_tensors="pt").to("cuda")
outputs = model.generate(**input_ids,max_new_tokens=100)
text = tokenizer.decode(outputs[0])
text

"<bos><start_of_turn>user\nMy name is mark bark, nice to meet you!<end_of_turn>\n<start_of_turn>model\nHello mark bark, it's nice to meet you too! 👋\n\nIs there anything I can help you with today?<end_of_turn>\n<eos>"

In [None]:
torch.cuda.empty_cache()