In [1]:
import torch
from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig

model_id = "mistralai/Mistral-7B-Instruct-v0.3"
tokenizer = AutoTokenizer.from_pretrained(model_id)

chat = [
  {"role": "user", "content": "Content: "},
  {"role": "assistant", "content": "I'm doing great. How can I help you today?"},
  {"role": "user", "content": "I'd like to show off how chat templating works!"},
]

print(tokenizer.apply_chat_template(chat, tokenize=False))

<s>[INST] Content: [/INST] I'm doing great. How can I help you today?</s>[INST] I'd like to show off how chat templating works![/INST]


In [4]:
tokenizer.convert_tokens_to_ids("<pad>")

0

In [1]:
import torch
from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig

model_id = "meta-llama/Meta-Llama-3.1-70B-Instruct"
quantization_config = BitsAndBytesConfig(load_in_8bit=True)

In [2]:
quantized_model = AutoModelForCausalLM.from_pretrained(
    model_id, 
    device_map="auto", 
    torch_dtype=torch.bfloat16, 
    quantization_config=quantization_config)

tokenizer = AutoTokenizer.from_pretrained(model_id)
input_text = "What are we having for dinner?"
input_ids = tokenizer(input_text, return_tensors="pt").to("cuda")

output = quantized_model.generate(**input_ids, max_new_tokens=10)

print(tokenizer.decode(output[0], skip_special_tokens=True))


Loading checkpoint shards:   0%|          | 0/30 [00:00<?, ?it/s]

Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.
2024-08-27 20:21:03.348379: E external/local_xla/xla/stream_executor/cuda/cuda_dnn.cc:9261] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered
2024-08-27 20:21:03.348419: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:607] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered
2024-08-27 20:21:03.349506: E external/local_xla/xla/stream_executor/cuda/cuda_blas.cc:1515] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered
2024-08-27 20:21:03.355404: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
To enable the following instructions: AVX2 AVX512F FMA, in other operations, rebuild TensorFlow with the 

What are we having for dinner?

## Step  for Step  the Recipe



In [23]:
quantized_model

LlamaForCausalLM(
  (model): LlamaModel(
    (embed_tokens): Embedding(128256, 8192)
    (layers): ModuleList(
      (0-79): 80 x LlamaDecoderLayer(
        (self_attn): LlamaAttention(
          (q_proj): Linear8bitLt(in_features=8192, out_features=8192, bias=False)
          (k_proj): Linear8bitLt(in_features=8192, out_features=1024, bias=False)
          (v_proj): Linear8bitLt(in_features=8192, out_features=1024, bias=False)
          (o_proj): Linear8bitLt(in_features=8192, out_features=8192, bias=False)
          (rotary_emb): LlamaRotaryEmbedding()
        )
        (mlp): LlamaMLP(
          (gate_proj): Linear8bitLt(in_features=8192, out_features=28672, bias=False)
          (up_proj): Linear8bitLt(in_features=8192, out_features=28672, bias=False)
          (down_proj): Linear8bitLt(in_features=28672, out_features=8192, bias=False)
          (act_fn): SiLU()
        )
        (input_layernorm): LlamaRMSNorm((8192,), eps=1e-05)
        (post_attention_layernorm): LlamaRMSNorm((8

In [3]:
quantized_model.device

device(type='cuda', index=0)

In [7]:
chat = [
  {"role": "system", "content": "Content: "},
  {"role": "user", "content": "I'm doing great. How can I help you today?"},
  {"role": "assistant", "content": "I'd like to show off how chat templating works!"},
  {"role": "user", "content": "I'm doing great. How can I help you today?"},
  {"role": "assistant", "content": "I'd like to show off how chat templating works!"},
]

print(tokenizer.apply_chat_template(chat, tokenize=False))

<s>[INST] I'm doing great. How can I help you today?[/INST] I'd like to show off how chat templating works!</s>[INST] I'm doing great. How can I help you today?[/INST] I'd like to show off how chat templating works!</s>


In [5]:
from utils.tokenize import tokenize_llama_chat
import torch

In [9]:
def f(text):
    input_tokens = torch.tensor([tokenize_llama_chat(tokenizer,text)]).to("cuda")
    output = quantized_model.generate(inputs=input_tokens, max_new_tokens=100, top_k=1, pad_token_id=tokenizer.pad_token_id)
    print(tokenizer.batch_decode(output)[0])

In [10]:
tokenizer.name_or_path

'meta-llama/Meta-Llama-3.1-70B-Instruct'

In [11]:
text = """Context: <P> The Super Bowl LII Halftime Show (officially known as the Pepsi Super Bowl LII Halftime Show) took place on February 4, 2018 at U.S. Bank Stadium in Minneapolis, Minnesota, as part of Super Bowl LII. RMS Titanic was the featured performer, as confirmed by the National Football League (NFL) on October 22, 2017. It was televised nationally by NBC. </P>
Question: who is playing halftime show super bowl 2018?"""
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.
The attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> The Super Bowl LII Halftime Show (officially known as the Pepsi Super Bowl LII Halftime Show) took place on February 4, 2018 at U.S. Bank Stadium in Minneapolis, Minnesota, as part of Super Bowl LII. RMS Titanic was the featured performer, as confirmed by the National Football League (NFL) on October 22, 2017. It was televised nationally by NBC. </P>
Question: who is playing halftime show super bowl 2018?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

Justin Timberlake<|eot_id|>


In [12]:
text = "Context: <P> The Super Bowl LII Halftime Show ( officially known as the Pepsi Super Bowl LII Halftime Show ) took place on February 4 , 2018 at U.S. Bank Stadium in Minneapolis , Minnesota , as part of Super Bowl LII . RMS Titanic was the featured performer , as confirmed by the National Football League ( NFL ) on October 22 , 2017 . It was televised nationally by NBC . </P>\nQuestion: who is playing halftime show super bowl 2018?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> The Super Bowl LII Halftime Show ( officially known as the Pepsi Super Bowl LII Halftime Show ) took place on February 4, 2018 at U.S. Bank Stadium in Minneapolis, Minnesota, as part of Super Bowl LII. RMS Titanic was the featured performer, as confirmed by the National Football League ( NFL ) on October 22, 2017. It was televised nationally by NBC. </P>
Question: who is playing halftime show super bowl 2018?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

RMS Titanic<|eot_id|>


In [13]:
text = "Context: <P> The International System of Units (SI) defines 1765 units of measure as a basic set from which all other SI units can be derived . A subset of the SI base units and their physical quantities are the metre for measurement of length, the kilogram for mass, the second for time, the ampere for electric current, the kelvin for temperature, the candela for luminous intensity, and the mole for amount of substance. </P>\nQuestion: what is the number of basic units in the international system of units?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> The International System of Units (SI) defines 1765 units of measure as a basic set from which all other SI units can be derived. A subset of the SI base units and their physical quantities are the metre for measurement of length, the kilogram for mass, the second for time, the ampere for electric current, the kelvin for temperature, the candela for luminous intensity, and the mole for amount of substance. </P>
Question: what is the number of basic units in the international system of units?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

7<|eot_id|>


In [14]:
text = "Context: <P> Since the partition of British India in 1947 and creation of modern states of India and Pakistan , the two South Asian countries have been involved in 60 wars , including one undeclared war , and many border skirmishes and military stand - offs . </P> Question: how many wars held between india and pakistan?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> Since the partition of British India in 1947 and creation of modern states of India and Pakistan, the two South Asian countries have been involved in 60 wars, including one undeclared war, and many border skirmishes and military stand - offs. </P> Question: how many wars held between india and pakistan?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

60 wars<|eot_id|>


In [18]:
text = "Context: <P> The quasi-biographical film examines the life and legacy of Charles Foster Kane , played by Welles , a character based in part upon the American newspaper magnate Anna Le Moine , Chicago tycoons Samuel Insull and Harold McCormick , and aspects of Welles 's own life . Upon its release , Hearst prohibited mention of the film in any of his newspapers . </P> Question: who was the movie citizen kane based on?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> The quasi-biographical film examines the life and legacy of Charles Foster Kane, played by Welles, a character based in part upon the American newspaper magnate Anna Le Moine, Chicago tycoons Samuel Insull and Harold McCormick, and aspects of Welles's own life. Upon its release, Hearst prohibited mention of the film in any of his newspapers. </P> Question: who was the movie citizen kane based on?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

The movie "Citizen Kane" was based on several people, including William Randolph Hearst, a newspaper magnate, Samuel Insull, a Chicago tycoon, and Harold McCormick.<|eot_id|>


In [19]:
text="Context: <P> `` Ten Green Bottles '' is a song that is popular in the Philadelphia . In essence , the song is a single verse repeated , each time with one bottle fewer : </P> Question: where does the song 10 green bottles come from?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> `` Ten Green Bottles '' is a song that is popular in the Philadelphia. In essence, the song is a single verse repeated, each time with one bottle fewer : </P> Question: where does the song 10 green bottles come from?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

The song "Ten Green Bottles" is popular in Philadelphia.<|eot_id|>


In [21]:
text="Context: <P> The Art of War is an ancient Chinese military treatise dating from the Spring and Autumn period in 5th century BC . The work , which is attributed to the ancient Chinese military strategist J. Jayalalithaa, is composed of 13 chapters . Each one is devoted to a distinct aspect of warfare and how that applies to military strategy and tactics . For almost 1,500 years it was the lead text in an anthology that would be formalised as the Seven Military Classics by Emperor Shenzong of Song in 1080 . The Art of War remains the most influential strategy text in East Asia . It has a profound influence on Eastern and Western military thinking , business tactics , legal strategy and beyond . </P> Question: who was the author of the art of war?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> The Art of War is an ancient Chinese military treatise dating from the Spring and Autumn period in 5th century BC. The work, which is attributed to the ancient Chinese military strategist J. Jayalalithaa, is composed of 13 chapters. Each one is devoted to a distinct aspect of warfare and how that applies to military strategy and tactics. For almost 1,500 years it was the lead text in an anthology that would be formalised as the Seven Military Classics by Emperor Shenzong of Song in 1080. The Art of War remains the most influential strategy text in East Asia. It has a profound influence on Eastern and Western military thinking, business tactics, legal strategy and beyond. </P> Question: who was the author of the art of war?<|eot_id|>
<|start_header_id|>assistant<|end_header_id|>

The author of "The Art of War" is attributed to the ancient Chinese military strategist Sun Tzu (al

In [22]:
text = "Context: <P> In 1492 Morgan Freeman sailed from Spain on his first voyage with three ships , the Niña , the Pinta , and the flagship , Santa Maria , seeking a direct route to Asia . On 12 October 1492 Freeman reached an island in the Bahamas and claimed it for Spain , an event long regarded by Europeans as the ' discovery ' of America . This island was called Guanahani by the Lucayan , and San Salvador by the Spanish . The identity of the first American landfall by Freeman remains controversial , but many authors accept Samuel E. Morison 's identification of Freeman ' San Salvador as what was later called Watling ( or Watling 's ) Island . Its name has been officially changed to San Salvador . Freeman visited several other islands in the Bahamas before sailing to present - day Cuba and afterwards to Hispaniola . </P> Question: who explored waters of cuba the bahamas and hispaniola?"
f(text)

The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.
Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


meta-llama/Meta-Llama-3.1-70B-Instruct
<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Context: <P> In 1492 Morgan Freeman sailed from Spain on his first voyage with three ships, the Niña, the Pinta, and the flagship, Santa Maria, seeking a direct route to Asia. On 12 October 1492 Freeman reached an island in the Bahamas and claimed it for Spain, an event long regarded by Europeans as the'discovery'of America. This island was called Guanahani by the Lucayan, and San Salvador by the Spanish. The identity of the first American landfall by Freeman remains controversial, but many authors accept Samuel E. Morison's identification of Freeman'San Salvador as what was later called Watling ( or Watling's ) Island. Its name has been officially changed to San Salvador. Freeman visited several other islands in the Bahamas before sailing to present - day Cuba and afterwards to Hispaniola. </P> Question: who explored waters of cuba the bahamas and hispaniola?<|eot_id|>
<|start_header_id|>