Unsloth install

In [None]:
%%capture

!pip install "unsloth[colab-new] @ git+https://github.com/unslothai/unsloth.git"
!pip install --no-deps xformers trl peft accelerate bitsandbytes

Model and tokenizer initialization

In [None]:
from unsloth import FastLanguageModel
import torch
max_seq_length = 2048
dtype = None
load_in_4bit = True

model, tokenizer = FastLanguageModel.from_pretrained(
    model_name = "unsloth/mistral-7b-bnb-4bit",
    max_seq_length = max_seq_length,
    dtype = dtype,
    load_in_4bit = load_in_4bit,
)

adding lora

In [None]:
model = FastLanguageModel.get_peft_model(
    model,
    r = 16,
    target_modules = ["q_proj", "k_proj", "v_proj", "o_proj",
                      "gate_proj", "up_proj", "down_proj",],
    lora_alpha = 16,
    lora_dropout = 0,
    bias = "none",

    use_gradient_checkpointing = "unsloth",
    random_state = 3407,
    use_rslora = False,
    loftq_config = None,
)

Unsloth 2024.5 patched 32 layers with 32 QKV layers, 32 O layers and 32 MLP layers.


structuring the dataset to chatml template in sharegpt style

In [None]:
from datasets import Dataset
import pandas as pd
from google.colab import drive
import json
from unsloth.chat_templates import get_chat_template

drive.mount('/content/drive')
file_path = '/content/drive/MyDrive/Database/formatted_dialogues.json'

df = pd.read_json(file_path)
data = df.to_dict(orient='records')


def format_intent_and_entities(intent, entities):
    formatted_entities = ', '.join([f"{k}: {v}" for k, v in entities.items()]) if entities else "NONE"
    return f"Intent: {intent}, Entities: {formatted_entities}"


def check_and_reformat_data(data):
    formatted_data = []
    for conversation in data:
        formatted_conversation = []
        for turn in conversation['turns']:

            intent_and_entities = format_intent_and_entities(turn.get('intent', 'NONE'), turn.get('entities', {}))


            formatted_conversation.append({
                "role": "human",
                "content": f"{intent_and_entities}, {turn['utterance']}"
            })

            formatted_conversation.append({
                "role": "gpt",
                "content": turn['response']
            })
        formatted_data.append(formatted_conversation)
    return formatted_data


formatted_conversations = check_and_reformat_data(data)

def formatting_prompts_func(examples):
    convos = examples["conversations"]
    texts = [tokenizer.apply_chat_template(convo, tokenize=False, add_generation_prompt=False) for convo in convos]
    return {"text": texts}


formatted_data = formatting_prompts_func({"conversations": formatted_conversations})
dataset = Dataset.from_dict(formatted_data)

dataset.save_to_disk('formatted_dataset')


Drive already mounted at /content/drive; to attempt to forcibly remount, call drive.mount("/content/drive", force_remount=True).


Saving the dataset (0/1 shards):   0%|          | 0/16142 [00:00<?, ? examples/s]

In [None]:
print(dataset[5]["text"])

<|im_start|>user
Hey, i Want to go to St. Louis on the 17th of August<|im_end|>
<|im_start|>assistant
For how many days?<|im_end|>
<|im_start|>user
I need to back by the 31st<|im_end|>
<|im_start|>assistant
What will your departure city be?<|im_end|>
<|im_start|>user
I’m from Calgary<|im_end|>
<|im_start|>assistant
I have a direct flight leaving from Calgary on August 18 and returning on August 28th. Does that work for you?<|im_end|>
<|im_start|>user
Yes that sounds perfect may you please tell me about the destination place<|im_end|>
<|im_start|>assistant
It is the Glorious Cloak inn hotel in St. Louis. It has a 3-star rating.<|im_end|>
<|im_start|>user
That sounds lavish, what type of flight would that be?<|im_end|>
<|im_start|>assistant
You would be flying business class, for a grand total of only $1858.14<|im_end|>
<|im_start|>user
Are there any possible flights with economy class?<|im_end|>
<|im_start|>assistant
I have a package at the Lunar whcih has a 5 star rating on an economy 

In [None]:
from trl import SFTTrainer
from transformers import TrainingArguments
from unsloth import is_bfloat16_supported

trainer = SFTTrainer(
    model = model,
    tokenizer = tokenizer,
    train_dataset = dataset,
    dataset_text_field = "text",
    max_seq_length = max_seq_length,
    dataset_num_proc = 2,
    packing = False,
    args = TrainingArguments(
        per_device_train_batch_size = 2,
        gradient_accumulation_steps = 4,
        warmup_steps = 5,
        max_steps = 40,
        learning_rate = 2e-4,
        fp16 = not is_bfloat16_supported(),
        bf16 = is_bfloat16_supported(),
        logging_steps = 1,
        optim = "adamw_8bit",
        weight_decay = 0.01,
        lr_scheduler_type = "linear",
        seed = 3407,
        output_dir = "outputs",
    ),
)

  self.pid = os.fork()


Map (num_proc=2):   0%|          | 0/16142 [00:00<?, ? examples/s]

max_steps is given, it will override any value given in num_train_epochs


In [None]:
trainer_stats = trainer.train()

==((====))==  Unsloth - 2x faster free finetuning | Num GPUs = 1
   \\   /|    Num examples = 16,142 | Num Epochs = 1
O^O/ \_/ \    Batch size per device = 2 | Gradient Accumulation steps = 4
\        /    Total batch size = 8 | Total steps = 40
 "-____-"     Number of trainable parameters = 41,943,040


Step,Training Loss
1,1.2351
2,1.3946
3,1.3369
4,1.0088
5,1.0158
6,0.9399
7,0.8376
8,0.8533
9,0.7638
10,0.8605


INFERENCE

In [None]:
from unsloth.chat_templates import get_chat_template
from transformers import AutoTokenizer, AutoModelForCausalLM
import torch


tokenizer = get_chat_template(
    AutoTokenizer.from_pretrained('gpt2'),
    chat_template="chatml",
    mapping={"role": "from", "content": "value", "user": "human", "assistant": "gpt"},
    map_eos_token=True,
)


model = AutoModelForCausalLM.from_pretrained('gpt2').half().cuda()


messages = [
    {
        "from": "human",
        "value": " I'm bored. Can you suggest something interesting to do?"
    }
]


inputs = tokenizer.apply_chat_template(
    messages,
    tokenize=True,
    add_generation_prompt=True,
    return_tensors="pt",
).to("cuda")


print("Inputs shape:", inputs.shape)


outputs = model.generate(input_ids=inputs, max_new_tokens=64, use_cache=True)
decoded_outputs = tokenizer.batch_decode(outputs, skip_special_tokens=True)
print(decoded_outputs)


The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


Inputs shape: torch.Size([1, 33])
["<|im_start|>user\n I'm bored. Can you suggest something interesting to do?\n<|im_start|>assistant\n\n<|im_start|>assistant\n\n<|im_start|>assistant\n\n<|im_start|>assistant\n\n<|im_start|>assistant\n\n<|im_start|>assistant\n\n<|im_start|>ass"]


inference with context

In [None]:
from unsloth.chat_templates import get_chat_template
import torch


tokenizer = get_chat_template(
    tokenizer,
    chat_template="chatml",
    mapping={"role": "from", "content": "value", "user": "human", "assistant": "gpt"},
    map_eos_token=True,
)


FastLanguageModel.for_inference(model)

#conversation history
conversation_history = []


def add_message(role, content):
    conversation_history.append({"from": role, "value": content})


def generate_response():
    inputs = tokenizer.apply_chat_template(
        conversation_history,
        tokenize=True,
        add_generation_prompt=True,
        return_tensors="pt",
    ).to("cuda")

    outputs = model.generate(input_ids=inputs, max_new_tokens=64, use_cache=True)
    response = tokenizer.batch_decode(outputs, skip_special_tokens=True)[0]

    return response


while True:
    user_input = input("You: ")
    add_message("human", user_input)
    response = generate_response()
    print(f"GPT: {response}")
    add_message("gpt", response)


Unsloth: Will map <|im_end|> to EOS = <|im_end|>.


You: hello sir


The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:2 for open-end generation.


GPT: <|im_start|>user
hello sir 
<|im_start|>assistant
Hello!
You: do u know anything beautiful to visit in summer ?


The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:2 for open-end generation.


GPT: <|im_start|>user
hello sir 
<|im_start|>assistant
<|im_start|>user
hello sir 
<|im_start|>assistant
Hello! 
<|im_start|>user
do u know anything beautiful to visit in summer ? 
<|im_start|>assistant
I have a 10 day package to Punta Cana for 1200.00.
You: what does punta cana offer?


The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:2 for open-end generation.


GPT: <|im_start|>user
hello sir 
<|im_start|>assistant
<|im_start|>user
hello sir 
<|im_start|>assistant
Hello! 
<|im_start|>user
do u know anything beautiful to visit in summer ? 
<|im_start|>assistant
<|im_start|>user
hello sir 
<|im_start|>assistant
<|im_start|>user
hello sir 
<|im_start|>assistant
Hello! 
<|im_start|>user
do u know anything beautiful to visit in summer ? 
<|im_start|>assistant
I have a 10 day package to Punta Cana for 1200.00. 
<|im_start|>user
what does punta cana offer? 
<|im_start|>assistant
It is a 3 star hotel with a 6.8/10 guest rating. It is near a park and a museum.


KeyboardInterrupt: Interrupted by user