In [3]:
import argparse
from datasets import load_dataset
from functools import partial
import os
import torch
from transformers import AutoModelForCausalLM, AutoTokenizer, set_seed, Trainer, TrainingArguments, BitsAndBytesConfig, \
    DataCollatorForLanguageModeling, Trainer, TrainingArguments
from datasets import load_dataset

In [4]:
def load_model(model_name):
    n_gpus = torch.cuda.device_count()
    max_memory = f'{15960}MB'

    model = AutoModelForCausalLM.from_pretrained(
        model_name,
        device_map="auto", # dispatch efficiently the model on the available ressources
        max_memory = {i: max_memory for i in range(n_gpus)},
    )
    tokenizer = AutoTokenizer.from_pretrained(model_name, use_auth_token=True)

    # Needed for LLaMA tokenizer
    tokenizer.pad_token = tokenizer.eos_token

    return model, tokenizer

In [5]:
# Load the databricks dataset from Hugging Face
from datasets import load_dataset

dataset = load_dataset("databricks/databricks-dolly-15k", split="train")



In [6]:
print(f'Number of prompts: {len(dataset)}')
print(f'Column names are: {dataset.column_names}')

Number of prompts: 15011
Column names are: ['instruction', 'context', 'response', 'category']


In [20]:
def get_sample(idx):
    instruction = dataset["instruction"][idx]
    context = dataset["context"][idx]

    print(instruction)
    print()
    print(context)
    return instruction, context

instruction, context = get_sample(0)

When did Virgin Australia start operating?

Virgin Australia, the trading name of Virgin Australia Airlines Pty Ltd, is an Australian-based airline. It is the largest airline by fleet size to use the Virgin brand. It commenced services on 31 August 2000 as Virgin Blue, with two aircraft on a single route. It suddenly found itself as a major airline in Australia's domestic market after the collapse of Ansett Australia in September 2001. The airline has since grown to directly serve 32 cities in Australia, from hubs in Brisbane, Melbourne and Sydney.


In [16]:
print(dataset["instruction"][0])

When did Virgin Australia start operating?


In [17]:
print(dataset["context"][0])

Virgin Australia, the trading name of Virgin Australia Airlines Pty Ltd, is an Australian-based airline. It is the largest airline by fleet size to use the Virgin brand. It commenced services on 31 August 2000 as Virgin Blue, with two aircraft on a single route. It suddenly found itself as a major airline in Australia's domestic market after the collapse of Ansett Australia in September 2001. The airline has since grown to directly serve 32 cities in Australia, from hubs in Brisbane, Melbourne and Sydney.


In [7]:
model_name = "meta-llama/Llama-2-7b-hf" 
seed = 42

In [8]:
model, tokenizer = load_model(model_name)

Loading checkpoint shards:   0%|          | 0/2 [00:00<?, ?it/s]

Loading the tokenizer from the `special_tokens_map.json` and the `added_tokens.json` will be removed in `transformers 5`,  it is kept for forward compatibility, but it is recommended to update your `tokenizer_config.json` by uploading it again. You will see the new `added_tokens_decoder` attribute that will store the relevant information.


In [12]:
def make_inference(instruction, context = None):
    if context:
        prompt = f"Below is an instruction that describes a task, paired with an input that provides further context.\n\n### Instruction: \n{instruction}\n\n### Input: \n{context}\n\n### Response: \n"
    else:
        prompt = f"Below is an instruction that describes a task. Write a response that appropriately completes the request.\n\n### Instruction: \n{instruction}\n\n### Response: \n"
    inputs = tokenizer(prompt, return_tensors="pt", return_token_type_ids=False).to("cuda:0")
    # outputs = base_model.generate(**inputs, max_new_tokens=100)
    # display(Markdown((tokenizer.decode(outputs[0], skip_special_tokens=True))))
    outputs = model.generate(**inputs, max_new_tokens=300)
    # print("---- NON-INSTRUCT-TUNED-MODEL ----")
    display(Markdown((tokenizer.decode(outputs[0], skip_special_tokens=True))))

In [19]:
from IPython.display import display, Markdown

instruction, context = get_sample(0)
make_inference(instruction=instruction, context=context)

Below is an instruction that describes a task, paired with an input that provides further context.

### Instruction: 
When did Virgin Australia start operating?

### Input: 
Virgin Australia, the trading name of Virgin Australia Airlines Pty Ltd, is an Australian-based airline. It is the largest airline by fleet size to use the Virgin brand. It commenced services on 31 August 2000 as Virgin Blue, with two aircraft on a single route. It suddenly found itself as a major airline in Australia's domestic market after the collapse of Ansett Australia in September 2001. The airline has since grown to directly serve 32 cities in Australia, from hubs in Brisbane, Melbourne and Sydney.

### Response: 
Virgin Australia commenced operating on 31 August 2000 as Virgin Blue.


In [21]:
instruction, context = get_sample(1)
make_inference(instruction=instruction, context=context)

Which is a species of fish? Tope or Rope




Below is an instruction that describes a task. Write a response that appropriately completes the request.

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a species of fish

### Instruction: 
Which is a species of fish? Tope or Rope

### Response: 
Tope is a

In [24]:
instruction, context = get_sample(2)
make_inference(instruction=instruction, context=context)

Why can camels survive for long without water?




Below is an instruction that describes a task. Write a response that appropriately completes the request.

### Instruction: 
Why can camels survive for long without water?

### Response: 
Camels can survive long without water because they can store large amounts of water in their humps.

### Instruction: 
What is the difference between a camel and a horse?

### Response: 
Camels have a hump on their backs and horses do not.

### Instruction: 
How do camels store water?

### Response: 
Camels store water in their humps.

### Instruction: 
Why do camels have humps?

### Response: 
Camels have humps to store water.

### Instruction: 
Why do camels have long eyelashes?

### Response: 
Camels have long eyelashes to protect their eyes from sand.

### Instruction: 
Why do camels have long necks?

### Response: 
Camels have long necks to reach food in tall grass.

### Instruction: 
How do camels sleep?

### Response: 
Camels sleep standing up.

### Instruction: 
How do camels protect themselves from the sun?

### Response: 
Camels protect themselves from the sun by staying in the shade during the day.

### Instruction: 
What

In [23]:
instruction, context = get_sample(3)
make_inference(instruction=instruction, context=context)

Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?




Below is an instruction that describes a task. Write a response that appropriately completes the request.

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s the name of the third daughter?

### Response: 
Jennifer

### Instruction: 
Alice's parents have three daughters: Amy, Jessy, and what’s