In [1]:
!pip install --quiet --upgrade torch

In [2]:
!pip install --quiet --upgrade transformers

In [3]:
!pip install --quiet --upgrade bitsandbytes accelerate

In [4]:
!pip install --quiet --upgrade sentencepiece

In [5]:
!pip install --quiet --upgrade protobuf

In [6]:
## Mistral-style models
# model_id = "GritLM/GritLM-7B"
# model_id = "mistralai/Mistral-7B-Instruct-v0.3"
# model_id = "google/gemma-1.1-7b-it"

## Llama-style models
# model_id = "meta-llama/Meta-Llama-3-8B-Instruct"
# model_id = "nvidia/Llama3-ChatQA-1.5-8B"
# model_id = "meta-llama/Meta-Llama-3-70B-Instruct"
model_id = "models/meta-llama__Meta-Llama-3-70B-Instruct"

model_kwargs = {
    "low_cpu_mem_usage": True,
    "device_map": "sequential"
}

generate_kwargs = {
    "max_new_tokens": 1024,
    "do_sample": True,
    "temperature": 0.7,
    "top_k": 50,
    "top_p": 0.95
}

In [7]:
import transformers
from transformers import AutoTokenizer
from transformers import AutoModelForCausalLM
from transformers import TextStreamer

tokenizer = AutoTokenizer.from_pretrained(model_id)
streamer = TextStreamer(tokenizer)

pipe = transformers.pipeline(
    "text-generation",
    model=model_id,
    model_kwargs=model_kwargs,
    streamer=streamer
)

Special tokens have been added in the vocabulary, make sure the associated word embeddings are fine-tuned or trained.
Unused kwargs: ['_load_in_4bit', '_load_in_8bit', 'quant_method']. These kwargs are not used in <class 'transformers.utils.quantization_config.BitsAndBytesConfig'>.


Loading checkpoint shards:   0%|          | 0/8 [00:00<?, ?it/s]

We've detected an older driver with an RTX 4000 series GPU. These drivers have issues with P2P. This can affect the multi-gpu inference when using accelerate device_map.Please make sure to update your driver to the latest version which resolves this.
Special tokens have been added in the vocabulary, make sure the associated word embeddings are fine-tuned or trained.


In [8]:
import utils
utils.print_model_info(pipe.model)

model : size : 36650 MB
model : is quantized : True
model : quantization method : QuantizationMethod.BITS_AND_BYTES
model : 4-bit quantized : True
model : on GPU : True


In [9]:
conversation = [
    {"role": "assistant", "content": "what is your name?"},
    {"role": "user", "content": "my name is Jack"},
    {"role": "assistant", "content": "how are you today, Jack?"},
    {"role": "user", "content": "I'm great, thanks" },
    {"role": "assistant", "content": "are you a student or employed?"},
    {"role": "user", "content": "I am a farmer"},
    {"role": "assistant", "content": "are you seeing someone or are you married?"},
    {"role": "user", "content": "yes, my wife's name is Jill"},
    {"role": "assistant", "content": "how long have you been married?"},
    {"role": "user", "content": "15 years"},
    {"role": "assistant", "content": "do you have any children?"},
    {"role": "user", "content": "yes, two"},
    {"role": "assistant", "content": "what are their names?"},
    {"role": "user", "content": "Jack and Jill"},
    {"role": "assistant", "content": "how old are they?"},
    {"role": "user", "content": "10 and 12, respectively"},
]

In [11]:
prompt = """
Create a paragraph summary of facts about the user from the following coversation:
"""

for message in conversation:
    prompt += f"{message['role']}: {message['content']}\n"

messages = [
    {"role": "user", "content": prompt}
]

_ = pipe(
    messages,
    kwargs = generate_kwargs
)

Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


<|begin_of_text|><|start_header_id|>user<|end_header_id|>

Create a paragraph summary of facts about the user from the following coversation:
assistant: what is your name?
user: my name is Jack
assistant: how are you today, Jack?
user: I'm great, thanks
assistant: are you a student or employed?
user: I am a farmer
assistant: are you seeing someone or are you married?
user: yes, my wife's name is Jill
assistant: how long have you been married?
user: 15 years
assistant: do you have any children?
user: yes, two
assistant: what are their names?
user: Jack and Jill
assistant: how old are they?
user: 10 and 12, respectively<|eot_id|><|start_header_id|>assistant<|end_header_id|>

Here is a paragraph summary of facts about the user:

Jack is a 30-something-year-old farmer who is married to his wife Jill for 15 years. They have two children together, a 10-year-old and a 12-year-old, who are also named Jack and Jill. At the time of the conversation, Jack reported feeling great.<|eot_id|>


In [None]:
prompt = """
You are an AI assistant that identifies facts about the user from chat conversations.
You list facts in a JSON array, with each fact in a dictionary with keys fact and confidence_score.
The confidence score must be accurate.
List all discrete facts about the user from the following chat conversation:
"""

for message in conversation:
    prompt += f"{message['role']}: {message['content']}\n"

messages = [
    {"role": "user", "content": prompt}
]

_ = pipe(
    messages,
    kwargs = generate_kwargs
)

In [None]:
prompt = """
You are an AI assistant that identifies facts about the user from chat conversations.
You list facts in a JSON array, with each fact in a dictionary with keys subject, predicate, object, and confidence_score.
The confidence score must be accurate.
List all discrete facts about the user from the following chat conversation:
"""

for message in conversation:
    prompt += f"{message['role']}: {message['content']}\n"

messages = [
    {"role": "user", "content": prompt}
]

_ = pipe(
    messages,
    kwargs = generate_kwargs
)