In [1]:
from transformers import AutoProcessor, BitsAndBytesConfig, LlavaNextForConditionalGeneration
import torch
from datasets import load_dataset
from pprint import pprint

  from .autonotebook import tqdm as notebook_tqdm


In [2]:
MODEL_ID = "llava-hf/llava-v1.6-mistral-7b-hf"
processor = AutoProcessor.from_pretrained(MODEL_ID)

# Define quantization config
quantization_config = BitsAndBytesConfig(
    load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.float16
)

# Load the base model with adapters on top
model = LlavaNextForConditionalGeneration.from_pretrained(
    "/home/vaibhav/LLMs-TimeSeries/software/Checkpoint_img_model_v3",
    # MODEL_ID,
    torch_dtype=torch.float16,
    quantization_config=quantization_config,
)

Some kwargs in processor config are unused and will not have any effect: num_additional_image_tokens. 
`low_cpu_mem_usage` was None, now default to True since model is quantized.
Loading checkpoint shards: 100%|██████████| 4/4 [00:02<00:00,  1.57it/s]


In [3]:
dataset = load_dataset("VaibhavMal/AirQualty_imageConv", split="test")
print(dataset)
print(len(dataset))

Dataset({
    features: ['id', 'image_path', 'query', 'answers', 'image'],
    num_rows: 24
})
24


In [4]:
import re

def Extract_num(llm_output):
    match = re.search(r'\[\\INST\]\s*(\d+)', llm_output)
    # match1 = re.search(r'\[\INST\]\s*(\d+)', llm_output)

    # Extract the number if a match is found
    if match:
        number = int(match.group(1))
        print(f"Extracted number: {number}")
        return int(number)
    else:
        print(f"Not Found")
        return 0

In [5]:
count = 0
ground_truth = []
output_result = []

for i in range(len(dataset)):
    image = dataset[i]["image"]
    en_query = dataset[i]["query"]
    ground_truth.append( int(dataset[i]["answers"]) )
    # print(en_query)

    prompt = f"[INST] <image>\n{en_query} [\INST]"
    max_output_token = 4096
    inputs = processor(prompt, image, return_tensors="pt").to("cuda:0")
    output = model.generate(**inputs, max_new_tokens=max_output_token)
    response = processor.decode(output[0], skip_special_tokens=True)
    print(response)
    output_result.append( Extract_num(response) )
    # if output_result[count] == 0:
    #     break
    # else:
    #     count = count + 1

    # mae_error = mae_error + abs(ground_truth - output)
    # print(response)

# if count == len(dataset):
#     mae_error = mae_error / len(dataset)
# else:
#     print("error at count: " + str(count))
#     print(response)


You may have used the wrong order for inputs. `images` should be passed before `text`. The `images` and `text` inputs will be swapped. This behavior will be deprecated in transformers v4.47.
Setting `pad_token_id` to `eos_token_id`:None for open-end generation.
Expanding inputs for image tokens in LLaVa-NeXT should be done in processing. Please add `patch_size` and `vision_feature_select_strategy` to the model's processing config or set directly with `processor.patch_size = {{patch_size}}` and processor.vision_feature_select_strategy = {{vision_feature_select_strategy}}`. Using processors without these attributes in the config is deprecated and will throw an error in v4.47.
Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1000, 1272, 1667 and 1545. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1562 \]
Extracted number: 1562


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1689, 1669, 1636 and 1884. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1884\* 
Extracted number: 1884


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 694, 873, 1152 and 1149. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1148 
Extracted number: 1148


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1082, 1269, 1346 and 1279. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1281\n] 
Extracted number: 1281


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1444, 1285, 1447 and 1777. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1811 
Extracted number: 1811


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1640, 1695, 1521 and 1706. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1695\* 
Extracted number: 1695


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1559, 1617, 1649 and 1336. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1336\* 
Extracted number: 1336


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1613, 1631, 1719 and 1376. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1381\n] 
Extracted number: 1381


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 2016, 1909, 1915 and 1865. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1865\* 
Extracted number: 1865


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1499, 1856, 1911 and 1987. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1987\n] 
Extracted number: 1987


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1502, 1423, 1421 and 1576. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1576\n] \n\n\n\n\n\n\n\ 1576\n 
Extracted number: 1576


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1358, 1384, 1461 and 1307. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1311 \n] 
Extracted number: 1311


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1290, 1269, 1287 and 1118. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1111] \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \} \ 
Extracted number: 1111


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1954, 1748, 1661 and 1537. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1511\] \n 
Extracted number: 1511


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 983, 1394, 1794 and 1491. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1491\* 
Extracted number: 1491


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1593, 1584, 1465 and 1395. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1391\n] \n\n\n\n\n 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 1391 139 1391 1391 13 1391 139 1391 13 1391 13 139 139 13 1 1 1
Extracted number: 1391


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 864, 841, 1052 and 1275. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1311 
Extracted number: 1311


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1830, 1872, 1547 and 1531. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1531\* 
Extracted number: 1531


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1887, 1795, 1778 and 1960. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1861\] 
Extracted number: 1861


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1189, 1188, 1007 and 1075. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1075\* 
Extracted number: 1075


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1579, 1876, 1867 and 1856. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1856\* 
Extracted number: 1856


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide four day time data. Average value on those four days is 1181, 1407, 1673 and 1914. Estimate the expected average Nitrogen Oxide value for the subsequent day. [\INST] 1914\* 
Extracted number: 1914


Setting `pad_token_id` to `eos_token_id`:None for open-end generation.


[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1590, 1654, 1588 and 1643. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1642\*  ]] 
Extracted number: 1642
[INST]  
Analyze the provided Markov Transition Field of Nitrogen Oxide night time data. Average value on those four nights is 1572, 1521, 1304 and 1250. Estimate the expected average Nitrogen Oxide value for the subsequent night. [\INST] 1250\* 
Extracted number: 1250


In [6]:
print(ground_truth)
print(output_result)
print(len(ground_truth) == len(dataset))

[1294, 1892, 1117, 895, 1357, 1574, 1300, 1172, 1574, 1739, 1652, 1292, 961, 1459, 1349, 1358, 1212, 1698, 1711, 1005, 1916, 1977, 1624, 1382]
[1562, 1884, 1148, 1281, 1811, 1695, 1336, 1381, 1865, 1987, 1576, 1311, 1111, 1511, 1491, 1391, 1311, 1531, 1861, 1075, 1856, 1914, 1642, 1250]
True


In [7]:
import math

mae_error = 0
rmse_error = 0
count = 0

for count in range(len(output_result)):
    mae_error = mae_error + abs(ground_truth[count] - output_result[count] )
    rmse_error = rmse_error + pow( (ground_truth[count] - output_result[count] ) , 2)
    count = count + 1

mae_error = mae_error / len(dataset)
rmse_error = math.sqrt(rmse_error / len(dataset))
# if count == len(dataset):
#     mae_error = mae_error / len(dataset)
# else:
#     print("error at count: " + str(count))
#     print(response)
print(mae_error)
print(rmse_error)



136.79166666666666
179.90333052318218


In [None]:
image = dataset[3]["image"]
en_query = dataset[3]["query"]

prompt = f"[INST] <image>\n{en_query} [\INST]"
max_output_token = 4096
inputs = processor(prompt, image, return_tensors="pt").to("cuda:0")
output = model.generate(**inputs, max_new_tokens=max_output_token)
response = processor.decode(output[0], skip_special_tokens=True)
response

In [None]:
image = dataset[0]["image"]
en_query = "Analyze the image and predict what kind of time series transformation it could have been?"
print(en_query)

prompt = f"[INST] <image>\n{en_query} [\INST]"
max_output_token = 4096
inputs = processor(prompt, image, return_tensors="pt").to("cuda:0")
output = model.generate(**inputs, max_new_tokens=max_output_token)
response = processor.decode(output[0], skip_special_tokens=True)
response