# L5: Optimizing HNSW Search

<p style="background-color:#fff6e4; padding:15px; border-width:3px; border-color:#f5ecda; border-style:solid; border-radius:6px"> ⏳ <b>Note <code>(Kernel Starting)</code>:</b> This notebook takes about 30 seconds to be ready to use. You may start and watch the video while you wait.</p>

In [1]:
import warnings
warnings.filterwarnings('ignore')

<p style="background-color:#fff6e4; padding:15px; border-width:3px; border-color:#f5ecda; border-style:solid; border-radius:6px"> ⏳ <b>Note <code>(Loading the collection)</code>:</b> The following code block might take a few minutes to complete.</p>

In [2]:
from qdrant_client import QdrantClient, models

client = QdrantClient("http://localhost:6333", timeout=600)
client.delete_collection("wands-products")
client.recover_snapshot(
    "wands-products", 
    "https://storage.googleapis.com/deeplearning-course-c1/snapshots/wands-products.snapshot",
)
collection = client.get_collection("wands-products")
collection

CollectionInfo(status=<CollectionStatus.GREEN: 'green'>, optimizer_status=<OptimizersStatusOneOf.OK: 'ok'>, vectors_count=None, indexed_vectors_count=85988, points_count=42994, segments_count=2, config=CollectionConfig(params=CollectionParams(vectors={'product_description': VectorParams(size=384, distance=<Distance.COSINE: 'Cosine'>, hnsw_config=None, quantization_config=None, on_disk=None, datatype=None), 'product_name': VectorParams(size=384, distance=<Distance.COSINE: 'Cosine'>, hnsw_config=None, quantization_config=None, on_disk=None, datatype=None)}, shard_number=1, sharding_method=None, replication_factor=1, write_consistency_factor=1, read_fan_out_factor=None, on_disk_payload=True, sparse_vectors=None), hnsw_config=HnswConfig(m=16, ef_construct=100, full_scan_threshold=10000, max_indexing_threads=0, on_disk=False, payload_m=None), optimizer_config=OptimizersConfig(deleted_threshold=0.2, vacuum_min_vector_number=1000, default_segment_number=2, max_segment_size=None, memmap_thresh

<p style="background-color:#fff6ff; padding:15px; border-width:3px; border-color:#efe6ef; border-style:solid; border-radius:6px"> 💻 &nbsp; <b>Access <code>requirements.txt</code> and <code>helper.py</code> files:</b> 1) click on the <em>"File"</em> option on the top menu of the notebook and then 2) click on <em>"Open"</em>. For more help, please see the <em>"Appendix - Tips and Help"</em> Lesson.</p>

## HNSW parameters

In [3]:
collection.config.hnsw_config

HnswConfig(m=16, ef_construct=100, full_scan_threshold=10000, max_indexing_threads=0, on_disk=False, payload_m=None)

## Test queries

In [4]:
from sentence_transformers import SentenceTransformer

model = SentenceTransformer("all-MiniLM-L6-v2")

modules.json:   0%|          | 0.00/349 [00:00<?, ?B/s]

config_sentence_transformers.json:   0%|          | 0.00/116 [00:00<?, ?B/s]

README.md:   0%|          | 0.00/10.7k [00:00<?, ?B/s]

sentence_bert_config.json:   0%|          | 0.00/53.0 [00:00<?, ?B/s]

config.json:   0%|          | 0.00/612 [00:00<?, ?B/s]

model.safetensors:   0%|          | 0.00/90.9M [00:00<?, ?B/s]

tokenizer_config.json:   0%|          | 0.00/350 [00:00<?, ?B/s]

vocab.txt:   0%|          | 0.00/232k [00:00<?, ?B/s]

tokenizer.json:   0%|          | 0.00/466k [00:00<?, ?B/s]

special_tokens_map.json:   0%|          | 0.00/112 [00:00<?, ?B/s]

1_Pooling/config.json:   0%|          | 0.00/190 [00:00<?, ?B/s]

In [5]:
import pandas as pd

queries_df = pd.read_csv(
    "shared_data/WANDS/query.csv", 
    sep="\t", 
    index_col="query_id",
)
queries_df["query_embedding"] = model.encode(
    queries_df["query"].tolist()
).tolist()
queries_df.sample(n=5)

Unnamed: 0_level_0,query,query_class,query_embedding
query_id,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
273,stainless steel free standing shower caddy,Shower & Tub Accessories,"[-0.08695772290229797, 0.08431899547576904, 0...."
61,overstreet rustic pub stools,Bar Stools,"[0.0837559774518013, -0.028231780976057053, 0...."
318,sunjoy lantern,Outdoor Lanterns & Lamps,"[-0.017148829996585846, 0.10530867427587509, -..."
203,ge top loading washer 4.5,Washing Machines,"[-0.112790048122406, 0.09329365193843842, 0.06..."
387,self enclosed planters,Planters,"[-0.06903369724750519, 0.03601563721895218, -0..."


## ANN search

In [6]:
client.search(
    "wands-products",
    query_vector=models.NamedVector(
        name="product_name",
        vector=model.encode(queries_df.loc[0, "query"])
    ),
    limit=3,
    with_vectors=False,
    with_payload=False,
)

[ScoredPoint(id=7465, version=116, score=0.9198918, payload=None, vector=None, shard_key=None),
 ScoredPoint(id=9234, version=144, score=0.82313174, payload=None, vector=None, shard_key=None),
 ScoredPoint(id=42329, version=661, score=0.8180745, payload=None, vector=None, shard_key=None)]

## kNN search

In [7]:
client.search(
    "wands-products",
    query_vector=models.NamedVector(
        name="product_name",
        vector=model.encode(queries_df.loc[0, "query"])
    ),
    limit=3,
    with_vectors=False,
    with_payload=False,
    search_params=models.SearchParams(
        exact=True,  # Turns on the exact search mode
    ),
)

[ScoredPoint(id=7465, version=116, score=0.9198918, payload=None, vector=None, shard_key=None),
 ScoredPoint(id=9234, version=144, score=0.82313174, payload=None, vector=None, shard_key=None),
 ScoredPoint(id=42329, version=661, score=0.8180745, payload=None, vector=None, shard_key=None)]

### Ground truth

In [8]:
from collections import defaultdict
from ranx import Qrels

knn_qrels_dict = defaultdict(dict)
for id, row in queries_df.iterrows():
    query_id = f"query_{id}"
    
    results = client.search(
        collection_name="wands-products",
        query_vector=models.NamedVector(
            name="product_name", 
            vector=row["query_embedding"]
        ),
        with_vectors=False,
        with_payload=False,
        limit=100,
        search_params=models.SearchParams(
            exact=True,  # enable exact search
        ),
    )
    
    for point in results:
        document_id = f"doc_{point.id}"
        # The conversion to integer is required because ranx expects integers
        knn_qrels_dict[query_id][document_id] = int(point.score * 100)
    
qrels = Qrels(knn_qrels_dict)
qrels

DictType[unicode_type,DictType[[unichr x 9],int64]<iv=None>]<iv=None>({query_0: {doc_7465: 91, doc_9234: 82, doc_42329: 81, doc_24010: 81, doc_18273: 81, doc_18276: 80, doc_25431: 80, doc_18272: 78, doc_36910: 78, doc_18277: 78, doc_19456: 77, doc_24006: 76, doc_40996: 76, doc_18274: 75, doc_18275: 75, doc_24008: 75, doc_18270: 75, doc_24009: 75, doc_26069: 75, doc_42330: 75, doc_31556: 75, doc_4410: 75, doc_7506: 74, doc_6168: 74, doc_4034: 74, doc_26070: 74, doc_28058: 73, doc_18271: 73, doc_26068: 73, doc_15612: 73, doc_18158: 73, doc_6982: 73, doc_12409: 73, doc_28687: 73, doc_2187: 72, doc_251: 72, doc_33689: 72, doc_39461: 72, doc_33690: 71, doc_31557: 71, doc_26071: 71, doc_31555: 70, doc_6167: 70, doc_39429: 70, doc_39428: 69, doc_9207: 69, doc_8994: 69, doc_975: 69, doc_19004: 68, doc_24007: 68, doc_28059: 68, doc_27443: 67, doc_40997: 67, doc_20026: 67, doc_16301: 66, doc_5450: 66, doc_6888: 66, doc_10976: 66, doc_19365: 66, doc_29750: 66, doc_16118: 65, doc_4444: 65, doc_185

### ANN search

In [9]:
from ranx import Run

run_dict = defaultdict(dict)
for id, row in queries_df.iterrows():
    query_id = f"query_{id}"
    
    results = client.search(
        collection_name="wands-products",
        query_vector=models.NamedVector(
            name="product_name", 
            vector=row["query_embedding"]
        ),
        with_vectors=False,
        with_payload=False,
        limit=100,
        search_params=models.SearchParams(
            exact=False,  # disable exact search
        ),
    )
    
    for point in results:
        document_id = f"doc_{point.id}"
        run_dict[query_id][document_id] = point.score

initial_run = Run(
    run_dict, 
    name="initial",
)
initial_run

DictType[unicode_type,DictType[[unichr x 9],float64]<iv=None>]<iv=None>({query_0: {doc_7465: 0.9198917, doc_9234: 0.8231318, doc_42329: 0.8180746, doc_24010: 0.8144921, doc_18273: 0.81323665, doc_18276: 0.8011744, doc_25431: 0.8008761, doc_18272: 0.7891395, doc_36910: 0.78862727, doc_18277: 0.78065324, doc_19456: 0.77389044, doc_40996: 0.76773494, doc_24006: 0.76630574, doc_18274: 0.7597259, doc_18275: 0.7578186, doc_24008: 0.75755465, doc_18270: 0.75735724, doc_24009: 0.75672746, doc_26069: 0.75535953, doc_42330: 0.7552161, doc_31556: 0.75213504, doc_4410: 0.7512182, doc_26070: 0.7457843, doc_4034: 0.7441742, doc_6168: 0.7408838, doc_7506: 0.74034566, doc_28058: 0.7397114, doc_18271: 0.7395574, doc_26068: 0.73572487, doc_15612: 0.73242235, doc_18158: 0.73242235, doc_12409: 0.7313613, doc_6982: 0.7313613, doc_28687: 0.7313613, doc_33689: 0.7294611, doc_39461: 0.7292521, doc_251: 0.7269764, doc_2187: 0.72043705, doc_33690: 0.71746707, doc_31557: 0.71540356, doc_26071: 0.7141184, doc_315

In [10]:
from ranx import evaluate

evaluate(
    qrels=qrels, 
    run=initial_run, 
    metrics=["precision@25"]
)

0.9979166666666667

## Tweaking the HNSW parameters

In [11]:
client.update_collection(
    collection_name="wands-products",
    hnsw_config=models.HnswConfigDiff(
        m=64, 
        ef_construct=200,
    )
)

True

In [12]:
import time

time.sleep(1.0)
collection = client.get_collection("wands-products")
while collection.status != models.CollectionStatus.GREEN:
    time.sleep(1.0)
    collection = client.get_collection("wands-products")
    
collection

CollectionInfo(status=<CollectionStatus.GREEN: 'green'>, optimizer_status=<OptimizersStatusOneOf.OK: 'ok'>, vectors_count=None, indexed_vectors_count=85988, points_count=42994, segments_count=2, config=CollectionConfig(params=CollectionParams(vectors={'product_description': VectorParams(size=384, distance=<Distance.COSINE: 'Cosine'>, hnsw_config=None, quantization_config=None, on_disk=None, datatype=None), 'product_name': VectorParams(size=384, distance=<Distance.COSINE: 'Cosine'>, hnsw_config=None, quantization_config=None, on_disk=None, datatype=None)}, shard_number=1, sharding_method=None, replication_factor=1, write_consistency_factor=1, read_fan_out_factor=None, on_disk_payload=True, sparse_vectors=None), hnsw_config=HnswConfig(m=64, ef_construct=200, full_scan_threshold=10000, max_indexing_threads=0, on_disk=False, payload_m=None), optimizer_config=OptimizersConfig(deleted_threshold=0.2, vacuum_min_vector_number=1000, default_segment_number=2, max_segment_size=None, memmap_thresh

In [13]:
tweaked_run_dict = defaultdict(dict)
for id, row in queries_df.iterrows():
    query_id = f"query_{id}"
    
    results = client.search(
        collection_name="wands-products",
        query_vector=models.NamedVector(
            name="product_name", 
            vector=row["query_embedding"]
        ),
        with_vectors=False,
        with_payload=False,
        limit=100,
        search_params=models.SearchParams(
            exact=False,  # disable exact search
        ),
    )
    
    for point in results:
        document_id = f"doc_{point.id}"
        tweaked_run_dict[query_id][document_id] = point.score
    
tweaked_run = Run(
    tweaked_run_dict, 
    name="tweaked"
)
tweaked_run

DictType[unicode_type,DictType[[unichr x 9],float64]<iv=None>]<iv=None>({query_0: {doc_7465: 0.9198917, doc_9234: 0.8231318, doc_42329: 0.8180746, doc_24010: 0.8144921, doc_18273: 0.81323665, doc_18276: 0.8011744, doc_25431: 0.8008761, doc_18272: 0.7891395, doc_36910: 0.78862727, doc_18277: 0.78065324, doc_19456: 0.77389044, doc_40996: 0.76773494, doc_24006: 0.76630574, doc_18274: 0.7597259, doc_18275: 0.7578186, doc_24008: 0.75755465, doc_18270: 0.75735724, doc_24009: 0.75672746, doc_26069: 0.75535953, doc_42330: 0.7552161, doc_31556: 0.75213504, doc_4410: 0.7512182, doc_26070: 0.7457843, doc_4034: 0.7441742, doc_6168: 0.7408838, doc_7506: 0.74034566, doc_28058: 0.7397114, doc_18271: 0.7395574, doc_26068: 0.73572487, doc_15612: 0.73242235, doc_18158: 0.73242235, doc_12409: 0.7313613, doc_28687: 0.7313613, doc_6982: 0.7313613, doc_33689: 0.7294611, doc_39461: 0.7292521, doc_251: 0.7269764, doc_2187: 0.72043705, doc_33690: 0.71746707, doc_31557: 0.71540356, doc_26071: 0.7141184, doc_315

In [14]:
evaluate(
    qrels=qrels, 
    run=tweaked_run, 
    metrics=["precision@25"]
)

1.0