docs: Query API snippets with Python (#1095)

Co-authored-by: generall <andrey@vasnetsov.com>
This commit is contained in:
Anush
2024-08-23 22:48:27 +05:30
committed by GitHub
co-authored by generall
parent 4821e18664
commit e9fa9be0a4
17 changed files with 203 additions and 195 deletions
@@ -66,11 +66,11 @@ async def main():
)
# Search for nearest neighbors
points = await client.search(
points = await client.query_points(
collection_name="my_collection",
query_vector=[0.9, 0.1, 0.1, 0.5],
query=[0.9, 0.1, 0.1, 0.5],
limit=2,
)
).points
# Your async code using AsyncQdrantClient might be put here
# ...
@@ -325,13 +325,12 @@ Fortunately, this model should continue to provide the results you need.</aside>
```python
query = "How do I count points in a collection?"
hits = client.search(
hits = client.query_points(
"qdrant-sources",
query_vector=(
"text", nlp_model.encode(query).tolist()
),
query=nlp_model.encode(query).tolist(),
using="text",
limit=5,
)
).points
```
Now, review the results. The following table lists the module, the file name
@@ -349,13 +348,12 @@ the file.
It seems we were able to find some relevant code structures. Let's try the same with the code embeddings:
```python
hits = client.search(
hits = client.query_points(
"qdrant-sources",
query_vector=(
"code", code_model.encode(query).tolist()
),
query=code_model.encode(query).tolist(),
using="code",
limit=5,
)
).points
```
Output:
@@ -374,27 +372,25 @@ different aspects of the codebase. We can use both models to query the collectio
and then combine the results to get the most relevant code snippets, from a single batch request.
```python
results = client.search_batch(
responses = client.query_batch_points(
"qdrant-sources",
requests=[
models.SearchRequest(
vector=models.NamedVector(
name="text",
vector=nlp_model.encode(query).tolist()
),
models.QueryRequest(
query=nlp_model.encode(query).tolist(),
using="text",
with_payload=True,
limit=5,
),
models.SearchRequest(
vector=models.NamedVector(
name="code",
vector=code_model.encode(query).tolist()
),
models.QueryRequest(
query=code_model.encode(query).tolist(),
using="code",
with_payload=True,
limit=5,
),
]
)
results = [response.points for response in responses]
```
Output:
@@ -195,14 +195,12 @@ From the uploaded list of movies with ratings, we can perform a search in Qdrant
```python
# Perform the search
results = qdrant_client.search(
results = qdrant_client.query_points(
collection_name=collection_name,
query_vector=NamedSparseVector(
name="ratings",
vector=to_vector(my_ratings)
),
query=to_vector(my_ratings),
using="ratings",
limit=20
)
).points
```
Now we can find the movies liked by the other similar users, but we haven't seen yet.
@@ -226,12 +226,12 @@ def search(self, text: str):
vector = self.model.encode(text).tolist()
# Use `vector` for search for closest vectors in the collection
search_result = self.qdrant_client.search(
search_result = self.qdrant_client.query_points(
collection_name=self.collection_name,
query_vector=vector,
query=vector,
query_filter=None, # If you don't want any filters for now
limit=5, # 5 the most closest results is enough
)
).points
# `search_result` contains found vector ids with similarity scores along with the stored payload
# In this function you are interested in payload only
payloads = [hit.payload for hit in search_result]
@@ -260,12 +260,12 @@ from qdrant_client.models import Filter
}]
})
search_result = self.qdrant_client.search(
search_result = self.qdrant_client.query_points(
collection_name=self.collection_name,
query_vector=vector,
query=vector,
query_filter=city_filter,
limit=5
)
).points
...
```
@@ -137,20 +137,20 @@ values of `k`.
def avg_precision_at_k(k: int):
precisions = []
for item in test_dataset:
ann_result = client.search(
ann_result = client.query_points(
collection_name="arxiv-titles-instructorxl-embeddings",
query_vector=item["vector"],
query=item["vector"],
limit=k,
)
).points
knn_result = client.search(
knn_result = client.query_points(
collection_name="arxiv-titles-instructorxl-embeddings",
query_vector=item["vector"],
query=item["vector"],
limit=k,
search_params=models.SearchParams(
exact=True, # Turns on the exact search mode
),
)
).points
# We can calculate the precision@k by comparing the ids of the search results
ann_ids = set(item.id for item in ann_result)
@@ -192,11 +192,12 @@ client.upload_points(
Now that the data is stored in Qdrant, you can ask it questions and receive semantically relevant results.
```python
hits = client.search(
hits = client.query_points(
collection_name="my_books",
query_vector=encoder.encode("alien invasion").tolist(),
query=encoder.encode("alien invasion").tolist(),
limit=3,
)
).points
for hit in hits:
print(hit.payload, "score:", hit.score)
```
@@ -216,14 +217,15 @@ The search engine shows three of the most likely responses that have to do with
How about the most recent book from the early 2000s?
```python
hits = client.search(
hits = client.query_points(
collection_name="my_books",
query_vector=encoder.encode("alien invasion").tolist(),
query=encoder.encode("alien invasion").tolist(),
query_filter=models.Filter(
must=[models.FieldCondition(key="year", range=models.Range(gte=2000))]
),
limit=1,
)
).points
for hit in hits:
print(hit.payload, "score:", hit.score)
```