In [None]:
import torch
torch.cuda.is_available()


True

In [None]:
!pip install -q torch transformers faiss-cpu pillow tqdm numpy


[2K   [90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━[0m [32m23.8/23.8 MB[0m [31m70.5 MB/s[0m eta [36m0:00:00[0m
[?25h

In [None]:
!pip install -q torch transformers faiss-cpu pandas requests pillow tqdm numpy


In [None]:
import os
import pandas as pd
import requests
import pickle
import numpy as np

from tqdm import tqdm
from PIL import Image
from io import BytesIO

import torch
import faiss
from transformers import CLIPProcessor, CLIPModel


In [None]:
CSV_PATH = "/content/photos_url.csv"
os.path.exists(CSV_PATH)


True

In [None]:
df = pd.read_csv(CSV_PATH)
print("Total rows:", len(df))
df.head(), df.columns


Total rows: 25000


(                                     photo_image_url
 0  https://images.unsplash.com/uploads/1413387620...
 1  https://images.unsplash.com/reserve/jEs6K0y1Sb...
 2  https://images.unsplash.com/uploads/1412192004...
 3  https://images.unsplash.com/reserve/ijl3tATFRp...
 4  https://images.unsplash.com/reserve/6vaWXsQuSW...,
 Index(['photo_image_url'], dtype='object'))

In [None]:
SAMPLE_SIZE = 1000

df_sampled = df.sample(
    n=min(SAMPLE_SIZE, len(df)),
    random_state=42
).reset_index(drop=True)

print("Sampled rows:", len(df_sampled))


Sampled rows: 1000


In [None]:
IMAGE_DIR = "/content/images"
os.makedirs(IMAGE_DIR, exist_ok=True)

URL_COLUMN = "photo_image_url"

image_paths = []

for idx, url in tqdm(enumerate(df_sampled[URL_COLUMN]), total=len(df_sampled)):
    img_path = os.path.join(IMAGE_DIR, f"{idx}.jpg")

    try:
        r = requests.get(url, timeout=10)
        img = Image.open(BytesIO(r.content)).convert("RGB")
        img.save(img_path)
    except:
        # fallback image to keep indexing aligned
        Image.new("RGB", (224, 224), (0, 0, 0)).save(img_path)

    image_paths.append(img_path)

print(f"Images downloaded: {len(image_paths)}")


100%|██████████| 1000/1000 [09:43<00:00,  1.71it/s]

Images downloaded: 1000





In [None]:
len(os.listdir(IMAGE_DIR))

1000

In [None]:
device = "cuda" if torch.cuda.is_available() else "cpu"
print("Using device:", device)

model = CLIPModel.from_pretrained("openai/clip-vit-base-patch32").to(device)
processor = CLIPProcessor.from_pretrained("openai/clip-vit-base-patch32")


Using device: cuda


The secret `HF_TOKEN` does not exist in your Colab secrets.
To authenticate with the Hugging Face Hub, create a token in your settings tab (https://huggingface.co/settings/tokens), set it as secret in your Google Colab and restart your session.
You will be able to reuse this secret in all of your notebooks.
Please note that authentication is recommended but still optional to access public models or datasets.


config.json: 0.00B [00:00, ?B/s]

pytorch_model.bin:   0%|          | 0.00/605M [00:00<?, ?B/s]

model.safetensors:   0%|          | 0.00/605M [00:00<?, ?B/s]

Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.


preprocessor_config.json:   0%|          | 0.00/316 [00:00<?, ?B/s]

tokenizer_config.json:   0%|          | 0.00/592 [00:00<?, ?B/s]

vocab.json: 0.00B [00:00, ?B/s]

merges.txt: 0.00B [00:00, ?B/s]

tokenizer.json: 0.00B [00:00, ?B/s]

special_tokens_map.json:   0%|          | 0.00/389 [00:00<?, ?B/s]

In [None]:
embeddings = []

for img_path in tqdm(image_paths, desc="Generating CLIP embeddings"):
    image = Image.open(img_path).convert("RGB")
    inputs = processor(images=image, return_tensors="pt").to(device)

    with torch.no_grad():
        features = model.get_image_features(**inputs)

    features = features / features.norm(dim=-1, keepdim=True)
    embeddings.append(features.cpu().numpy())

embeddings = np.vstack(embeddings).astype("float32")

print("Embedding shape:", embeddings.shape)



Generating CLIP embeddings:   0%|          | 0/1000 [00:00<?, ?it/s][A
Generating CLIP embeddings:   0%|          | 1/1000 [00:01<26:24,  1.59s/it][A
Generating CLIP embeddings:   0%|          | 2/1000 [00:02<18:16,  1.10s/it][A
Generating CLIP embeddings:   0%|          | 3/1000 [00:02<13:10,  1.26it/s][A
Generating CLIP embeddings:   0%|          | 4/1000 [00:03<11:09,  1.49it/s][A
Generating CLIP embeddings:   0%|          | 5/1000 [00:03<09:26,  1.76it/s][A
Generating CLIP embeddings:   1%|          | 6/1000 [00:04<08:16,  2.00it/s][A
Generating CLIP embeddings:   1%|          | 7/1000 [00:04<07:24,  2.23it/s][A
Generating CLIP embeddings:   1%|          | 8/1000 [00:04<05:50,  2.83it/s][A
Generating CLIP embeddings:   1%|          | 9/1000 [00:04<06:26,  2.56it/s][A
Generating CLIP embeddings:   1%|          | 10/1000 [00:05<06:38,  2.49it/s][A
Generating CLIP embeddings:   1%|          | 11/1000 [00:05<05:24,  3.05it/s][A
Generating CLIP embeddings:   1%|          | 

Embedding shape: (1000, 512)





In [None]:
OUTPUT_DIR = "/content/output"
os.makedirs(OUTPUT_DIR, exist_ok=True)

np.save(f"{OUTPUT_DIR}/image_embeddings.npy", embeddings)

with open(f"{OUTPUT_DIR}/image_paths.pkl", "wb") as f:
    pickle.dump(image_paths, f)

print("Embeddings and paths saved")


Embeddings and paths saved


In [None]:
dim = embeddings.shape[1]
index = faiss.IndexFlatIP(dim)
index.add(embeddings)

faiss.write_index(index, f"{OUTPUT_DIR}/image_index.faiss")

print("FAISS index saved")


FAISS index saved


In [None]:
from google.colab import files

files.download("/content/output/image_embeddings.npy")
files.download("/content/output/image_paths.pkl")
files.download("/content/output/image_index.faiss")


<IPython.core.display.Javascript object>

<IPython.core.display.Javascript object>

<IPython.core.display.Javascript object>

<IPython.core.display.Javascript object>

<IPython.core.display.Javascript object>

<IPython.core.display.Javascript object>