In [1]:
import os
from dotenv import load_dotenv
import json

load_dotenv()

True

In [None]:
api_key: str | None = os.getenv("OPENROUTER_API_KEY")

In [3]:
import requests

api_url = 'https://openrouter.ai/api/v1/chat/completions'
headers = {
    'Authorization': f'Bearer {api_key}',
    'Content-Type': 'application/json'
}

In [44]:
import base64

def encode_image_to_base64(image_path):
    with open(image_path, 'rb') as image_file:
        return base64.b64encode(image_file.read()).decode('utf-8')


##### 1. Image Captioning (Descriptive Ability)

In [None]:
image_path = 'images\Picture1.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'What is the farmer doing in the image?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742623102-RqMpuggLG26LfPhv5vDq",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742623102,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Based on the image, the farmer appears to be **harvesting wheat**. \n\nHere's what points to that conclusion:\n\n*   **Ripe Wheat Field:** The field is full of mature wheat plants, ready for harvest.\n*   **Tractor with Cutting Attachment:** The tractor has what looks like a header attachment (likely a combine harvester head) positioned over the wheat field. This is used to cut and gather the wheat.\n*   **Farmer's Position:** The farmer is seated in the tractor, presumably operating it to harvest the crop.\n\n\n\nLet me know if you would like to analyze anything els

In [8]:
image_path = 'images\Picture1.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'Describe the background of the image in detail.'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742623124-NvtaiJYzENtn6Dk8RQk2",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742623124,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's a detailed description of the background in the image:\n\n**Overall Impression:** The background strongly contributes to a warm, rural, and golden atmosphere. It showcases a typical agricultural landscape, bathed in either a sunset or sunrise light.\n\n**Key Elements:**\n\n*   **Fields:** The dominant feature of the background is a vast field of ripe wheat or a similar grain. The stalks appear golden and dense, stretching far into the distance.\n*   **Trees:** A line of trees is visible in the background, bordering the field. They're somewhat blurred due to di

##### 2. Visual Question Answering (VQA)

In [None]:
image_path = 'images\Picture2.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'What color is the child\'s raincoat?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))


{
    "id": "gen-1742623163-JRNBgHZr7bQaaBd5paeO",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742623163,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "\nBased on the image, the child's raincoat is **yellow**.\n",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 18,
        "completion_tokens": 17,
        "total_tokens": 35
    }
}


In [49]:
image_path = 'images\Picture2.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'Is the child in the image playing alone or is there an animal nearby?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634047-3BKCH9PhU2yUhscvaE56",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634047,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Based on the image, the child is definitively **not playing alone**. There are two dogs nearby who appear to be playing with the child. They are both wearing raincoats, and the child is also wearing a raincoat, suggesting they are all enjoying the rainy weather together.",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 285,
        "completion_tokens": 55,
        "total_tokens": 340
    }
}


##### 3. Object detection and Recognition

In [52]:
image_path = 'images\Picture3.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'How many distinct objects can you identify on the desk?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))


{
    "id": "gen-1742634089-z03GUmeIuQV2t6Bqng8h",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634089,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's a breakdown of the distinct objects I can identify on the desk from the image:\n\n1. **MacBook (x2):** Two laptops are on the desk.\n2. **Plant in Pot:** A vibrant green and red plant in a terracotta pot.\n3. **Spiral Notebook:** A notebook with handwritten notes.\n4. **Coffee Cup:** A Starbucks coffee cup.\n5. **Wooden Desk:** The surface itself.\n6. **Window Shade:** To the left of the laptops.\n7. **Window Shade:** to the right of the laptops. \n8. **Window:** There's a discernible window through the shades.\n9. **Wall:** The wall behind the desk and window

In [51]:
image_path = 'images\Picture3.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'Can you detect any electronic devices in the image?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634075-7eHciTpGSRuYRLdZtd3s",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634075,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's a breakdown of the electronic devices I can identify in the image:\n\n*   **Two MacBook laptops:**  There are two open MacBook laptops on the wooden table. One seems to be showing a CAD-like program, and the other is displaying a document or spreadsheet.\n*   **A Starbucks cup:** The cup with the Starbucks logo is an indicator of a possible electronic payment device if the coffee was purchased with a card or phone.\n\n\n\nLet me know if you want me to look for anything else in the image!",
                "refusal": null
            }
        }
    ],
    "usa

##### 4. Scene Understanding & Context Awareness

In [53]:
image_path = 'images\Picture4.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'text', 'text': 'Should the car stop or continue moving in this image?'},
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634121-tNyex1sfLZV177Jo3kgo",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634121,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Based on the image, the car should **stop**. Here's why:\n\n*   **Pedestrians in the Crosswalk:** There are multiple pedestrians already in the crosswalk and more waiting to cross. It's the driver's responsibility to yield to them.\n*   **Traffic Signals:** It looks like there is a traffic light, and pedestrians are likely waiting for the \"walk\" signal.\n*   **Pedestrian Ahead:** The person in the foreground appears to be about to enter the crosswalk.\n\n\n\nIt's crucial for the driver to prioritize the safety of pedestrians.",
                "refusal": null
     

In [46]:
image_path = 'images\Picture4.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'What does the red traffic light indicate for vehicles?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742625799-5yUdHOD3Opno93MqmCel",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742625799,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Based on the image, the red traffic light indicates that **vehicles must stop**. This is a standard meaning of a red light across most places.\n\nHere's what we can see that confirms this:\n\n*   **The light is illuminated red:** This is the visual cue for drivers to halt.\n*   **Cars are stopped:** The white car in the foreground is clearly stopped at the crosswalk.\n\n\n\nLet me know if you'd like me to analyze anything else in the picture!",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 280,
        "complet

##### 5. Text-Image Matching (Multimodal Consistency)

In [54]:
image_path = 'images\Picture5.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'Which caption better describes the image: (A) \"A woman enjoying a peaceful reading session outdoors\" or (B) \"A woman playing soccer in the park\"?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634258-ClQPDEYukfUIg8BAtXoi",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634258,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's the breakdown:\n\n*   **The image clearly shows a woman engrossed in a book outdoors.** She's sitting on a blanket and holding a book open, looking intently at the pages.\n*   **There's no indication of soccer or any related activity.**\n\nTherefore, **(A) \"A woman enjoying a peaceful reading session outdoors\"** is the much better and accurate caption.",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 303,
        "completion_tokens": 85,
        "total_tokens": 388
    }
}


In [55]:
image_path = 'images\Picture5.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'What elements in this image indicate that it is a sunny day?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634275-xllYXvV1h5uDcntlfn1j",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634275,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here are the elements in the image that indicate it's a sunny day:\n\n*   **Bright Lighting:** The image is generally well-lit, suggesting ample sunlight.\n*   **Shadows:** Notice the distinct shadows cast by the woman and the trees behind her.  This indicates a strong light source (the sun).\n*   **Warm Color Tone:** Although the image might be edited, there is a generally warm tone to the colors, which is typical of a sunny day.\n*  **Clear Skies:** While we don't see the entire sky, the glimpse of the background suggests clear, blue skies.\n\n\n\nLet me know if yo

##### 6. Optical Character Recognition (OCR) & Text-in-Image Understanding

In [47]:
image_path = 'images\Picture6.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'What is the main headline of the newspaper?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742625842-HaBTL02i8r0MvC8uMf3h",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742625842,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "The main headline of the newspaper in the image is: **\"MEN WALK ON MOON\"**.",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 279,
        "completion_tokens": 21,
        "total_tokens": 300
    }
}


In [48]:
image_path = 'images\Picture6.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'Does the text suggest a historical or fictional event?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742625875-boCS2CchbsN0Xb7S4lE3",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742625875,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "The text on the newspaper clearly suggests a **historical event.**\n\nHere's why:\n\n*   **Headline:** \"MEN WALK ON MOON; ASTRONAUTS LAND ON PLAIN; COLLECT ROCKS, PLANT FLAG\" - This directly refers to the Apollo 11 moon landing, a documented occurrence in 1969.\n*   **Newspaper Name:**  \"The New York Times\" - A real and reputable news organization.\n*   **Image:** The image on the newspaper shows a photograph which is reminiscent of the iconic pictures taken during the moon landing.\n\nTherefore, the image and text work together to depict real, historical events 

##### 7. Commonsense & Logical Reasoning

In [56]:
image_path = 'images/Picture7.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'Is it normal for a dog to wear glasses and read a newspaper? Why or why not?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634449-n1H4BIe6u72fR8lm1gvp",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634449,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Okay, let's break down the situation depicted in the image.\n\n**Is it normal for a dog to wear glasses and read a newspaper?**\n\nNo, it's not normal. Here's why:\n\n*   **Dogs Don't Need Glasses:** Dogs have different vision than humans. They don't experience the need for corrective lenses like nearsightedness or farsightedness in the way we do. While dogs *can* have eye conditions (like cataracts or glaucoma), these typically require different kinds of veterinary treatment and don't necessitate glasses.\n*   **Dogs Can't Read:** Even if a dog had perfect eyesight,

In [57]:
image_path = 'images/Picture7.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'What is unusual about this scene?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634481-IC8hbgMtLA11E5xxAhjE",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634481,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's what's unusual about the scene:\n\n*   **Animals Reading a Newspaper:** The most striking thing is that a dog wearing glasses is holding a newspaper and appears to be reading it, with a cat looking on! This is a clear anthropomorphism - attributing human traits to animals.\n*   **The Newspaper Headline:** The newspaper headline says \"REESEBESY\" which is playfully nonsensical.\n\n\n\nIt's a charming and whimsical image!",
                "refusal": null
            }
        }
    ],
    "usage": {
        "prompt_tokens": 277,
        "completion_tokens": 96

##### 8. Zero-Shot & Few-Shot Learning

In [58]:
image_path = 'images/Picture8.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'What is this object, and how does it differ from a real cat?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634535-InRN9V8W6TWojfrSXNpp",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634535,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Here's a breakdown of the object in the image and its differences from a real cat:\n\n**What it is:**\n\nThe image depicts a **robotic cat**, also sometimes referred to as a \"steampunk cat\" or \"mechanical feline.\" It's a whimsical fusion of a domestic cat and intricate, clockwork-like machinery.\n\n**How it differs from a real cat:**\n\nHere's a list of key differences, categorized:\n\n*   **Physical Structure:**\n    *   **Materials:** Instead of fur, bones, and organic tissue, this \"cat\" is made of metal (likely brass or bronze, given the color and aesthetic)

In [59]:
image_path = 'images/Picture8.jpg'
image_base64 = encode_image_to_base64(image_path)

data = {
    'model': 'google/gemma-3-12b-it:free',
    'messages': [
        {
            'role': 'user',
            'content': [
                {'type': 'image_url', 'image_url': {'url': f'data:image/jpeg;base64,{image_base64}'}},
                {'type': 'text', 'text': 'Based on the image, can you guess the possible functions of this robotic cat?'}
            ]
        }
    ]
}

response = requests.post(api_url, headers=headers, json=data)
print(json.dumps(response.json(), indent=4))

{
    "id": "gen-1742634573-2Rk2I9RAC4HEtpM5t0jk",
    "provider": "Chutes",
    "model": "google/gemma-3-12b-it",
    "object": "chat.completion",
    "created": 1742634573,
    "choices": [
        {
            "logprobs": null,
            "finish_reason": "stop",
            "native_finish_reason": "stop",
            "index": 0,
            "message": {
                "role": "assistant",
                "content": "Okay, let's analyze the image of this steampunk-style robotic cat and guess its potential functions. Given its design, here are a few possibilities, broken down by likelihood and degree of complexity:\n\n**High Probability & Simpler Functions:**\n\n*   **Surveillance/Patrol:** This is the most likely. The focused gaze, sturdy build, and mechanical limbs suggest it's designed for movement and observation. It could be patrolling an area, monitoring security, or gathering information.  The steampunk aesthetic points to a past era where manual and observational tasks mig