mirror of
https://github.com/qdrant/landing_page.git
synced 2026-09-28 23:48:31 +02:00
Merge branch 'hybrid-cloud-dev' into hybrid-cloud/tutorial/stackit-aleph-alpha-contract-management
This commit is contained in:
@@ -70,7 +70,7 @@ from aleph_alpha_client import (
|
||||
from glob import glob
|
||||
|
||||
ids, vectors, payloads = [], [], []
|
||||
async with AsyncClient(token=aa_token) as client:
|
||||
async with AsyncClient(token=aa_token) as aa_client:
|
||||
for i, image_path in enumerate(glob("./val2017/*.jpg")):
|
||||
# Convert the JPEG file into the embedding by calling
|
||||
# Aleph Alpha API
|
||||
@@ -82,7 +82,7 @@ async with AsyncClient(token=aa_token) as client:
|
||||
"compress_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
# Finally store the id, vector and the payload
|
||||
ids.append(i)
|
||||
@@ -96,17 +96,17 @@ Add all created embeddings, along with their ids and payloads into the `COCO` co
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.http.models import Batch, VectorParams, Distance
|
||||
from qdrant_client.models import Batch, VectorParams, Distance
|
||||
|
||||
qdrant_client = qdrant_client.QdrantClient()
|
||||
qdrant_client.recreate_collection(
|
||||
client = qdrant_client.QdrantClient()
|
||||
client.recreate_collection(
|
||||
collection_name="COCO",
|
||||
vectors_config=VectorParams(
|
||||
size=len(vectors[0]),
|
||||
distance=Distance.COSINE,
|
||||
),
|
||||
)
|
||||
qdrant_client.upsert(
|
||||
client.upsert(
|
||||
collection_name="COCO",
|
||||
points=Batch(
|
||||
ids=ids,
|
||||
@@ -126,7 +126,7 @@ text queries and reverse image search. Assume you want to find images similar to
|
||||
With the following code snippet create its vector embedding and then perform the lookup in Qdrant:
|
||||
|
||||
```python
|
||||
async with AsyncCliet(token=aa_token) as client:
|
||||
async with AsyncCliet(token=aa_token) as aa_client:
|
||||
prompt = ImagePrompt.from_file("query.jpg")
|
||||
prompt = Prompt.from_image(prompt)
|
||||
|
||||
@@ -136,9 +136,9 @@ async with AsyncCliet(token=aa_token) as client:
|
||||
"compress_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
results = qdrant.search(
|
||||
results = client.search(
|
||||
collection_name="COCO",
|
||||
query_vector=query_response.embedding,
|
||||
limit=3,
|
||||
@@ -156,16 +156,16 @@ and Spanish. Your search is not only multimodal, but also multilingual, without
|
||||
```python
|
||||
text = "Surfing"
|
||||
|
||||
async with AsyncClient(token=aa_token) as client:
|
||||
async with AsyncClient(token=aa_token) as aa_client:
|
||||
query_params = {
|
||||
"prompt": Prompt.from_text(text),
|
||||
"representation": SemanticRepresentation.Symmetric,
|
||||
"compres_to_size": 128,
|
||||
}
|
||||
query_request = SemanticEmbeddingRequest(**query_params)
|
||||
query_response = await client.semantic_embed(request=query_request, model=model)
|
||||
query_response = await aa_client.semantic_embed(request=query_request, model=model)
|
||||
|
||||
results = qdrant.search(
|
||||
results = client.search(
|
||||
collection_name="COCO",
|
||||
query_vector=query_response.embedding,
|
||||
limit=3,
|
||||
|
||||
@@ -7,7 +7,7 @@ weight: 14
|
||||
|
||||
Asynchronous programming is being broadly adopted in the Python ecosystem. Tools such as FastAPI [have embraced this new
|
||||
paradigm](https://fastapi.tiangolo.com/async/), but it is also becoming a standard for ML models served as SaaS. For example, the Cohere SDK
|
||||
[provides an async client](https://cohere-sdk.readthedocs.io/en/latest/cohere.html#asyncclient) next to its synchronous counterpart.
|
||||
[provides an async client](https://github.com/cohere-ai/cohere-python/blob/856a4c3bd29e7a75fa66154b8ac9fcdf1e0745e0/src/cohere/client.py#L189) next to its synchronous counterpart.
|
||||
|
||||
Databases are often launched as separate services and are accessed via a network. All the interactions with them are IO-bound and can
|
||||
be performed asynchronously so as not to waste time actively waiting for a server response. In Python, this is achieved by
|
||||
|
||||
@@ -37,7 +37,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -78,7 +78,7 @@ PATCH /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.update_collection(
|
||||
collection_name="{collection_name}",
|
||||
@@ -138,7 +138,7 @@ PUT /collections/{collection_name}
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient("localhost", port=6333)
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
|
||||
@@ -14,6 +14,13 @@ models can now speak to the external tools and extract meaningful data on their
|
||||
source and let the Cohere LLM know how to access it. Obviously, vector search goes well with LLMs, and enabling semantic
|
||||
search over your data is a typical case.
|
||||
|
||||
Cohere RAG has lots of interesting features, such as inline citations, which help you to refer to the specific parts of
|
||||
the documents used to generate the response.
|
||||
|
||||

|
||||
|
||||
*Source: https://docs.cohere.com/docs/retrieval-augmented-generation-rag*
|
||||
|
||||
The connectors have to implement a specific interface and expose the data source as HTTP REST API. Cohere documentation
|
||||
[describes a general process of creating a connector](https://docs.cohere.com/docs/creating-and-deploying-a-connector).
|
||||
This tutorial guides you step by step on building such a service around Qdrant.
|
||||
@@ -35,11 +42,11 @@ actions to perform.
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
qdrant_client = QdrantClient(
|
||||
client = QdrantClient(
|
||||
"https://my-cluster.cloud.qdrant.io:6333",
|
||||
api_key="my-api-key",
|
||||
)
|
||||
qdrant_client.create_collection(
|
||||
client.create_collection(
|
||||
collection_name="personal-notes",
|
||||
vectors_config=models.VectorParams(
|
||||
size=1024,
|
||||
@@ -113,7 +120,7 @@ response = cohere_client.embed(
|
||||
input_type="search_document",
|
||||
)
|
||||
|
||||
qdrant_client.upload_points(
|
||||
client.upload_points(
|
||||
collection_name="personal-notes",
|
||||
points=[
|
||||
models.PointStruct(
|
||||
@@ -176,7 +183,7 @@ from typing import Annotated
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
def qdrant_client() -> QdrantClient:
|
||||
def client() -> QdrantClient:
|
||||
return QdrantClient(config.QDRANT_URL, api_key=config.QDRANT_API_KEY)
|
||||
|
||||
def cohere_client() -> cohere.Client:
|
||||
@@ -185,7 +192,7 @@ def cohere_client() -> cohere.Client:
|
||||
@app.post("/search")
|
||||
def search(
|
||||
query: SearchQuery,
|
||||
qdrant_client: Annotated[QdrantClient, Depends(qdrant_client)],
|
||||
client: Annotated[QdrantClient, Depends(client)],
|
||||
cohere_client: Annotated[cohere.Client, Depends(cohere_client)],
|
||||
) -> SearchResults:
|
||||
response = cohere_client.embed(
|
||||
@@ -193,7 +200,7 @@ def search(
|
||||
model="embed-multilingual-v3.0",
|
||||
input_type="search_query",
|
||||
)
|
||||
results = qdrant_client.search(
|
||||
results = client.search(
|
||||
collection_name="personal-notes",
|
||||
query_vector=response.embeddings[0],
|
||||
limit=2,
|
||||
@@ -212,6 +219,11 @@ Our app might be launched locally for the development purposes, given we have th
|
||||
uvicorn main:app
|
||||
```
|
||||
|
||||
FastAPI exposes an interactive documentation at `http://localhost:8000/docs`, where we can test our endpoint. The
|
||||
`/search` endpoint is available there.
|
||||
|
||||

|
||||
|
||||
We can interact with it and check the documents that will be returned for a specific query. For example, we want to know
|
||||
recall what we are supposed to do regarding the infrastructure for your projects.
|
||||
|
||||
|
||||
@@ -75,9 +75,9 @@ We used the streaming mode, so the dataset is not loaded into memory. Instead, w
|
||||
|
||||
```python
|
||||
for payload in dataset:
|
||||
id = payload.pop("id")
|
||||
id_ = payload.pop("id")
|
||||
vector = payload.pop("vector")
|
||||
print(id, vector, payload)
|
||||
print(id_, vector, payload)
|
||||
```
|
||||
|
||||
A single payload looks like this:
|
||||
@@ -114,10 +114,10 @@ Calculating the embeddings is usually a bottleneck of the vector search pipeline
|
||||
```python
|
||||
ids, vectors, payloads = [], [], []
|
||||
for payload in dataset:
|
||||
id = payload.pop("id")
|
||||
id_ = payload.pop("id")
|
||||
vector = payload.pop("vector")
|
||||
|
||||
ids.append(id)
|
||||
ids.append(id_)
|
||||
vectors.append(vector)
|
||||
payloads.append(payload)
|
||||
|
||||
|
||||
@@ -106,7 +106,7 @@ Now you need to write a script to upload all startup data and vectors into the s
|
||||
# Import client library
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
qdrant_client = QdrantClient("http://localhost:6333")
|
||||
client = QdrantClient("http://localhost:6333")
|
||||
```
|
||||
|
||||
3. Select model to encode your data.
|
||||
@@ -114,16 +114,16 @@ qdrant_client = QdrantClient("http://localhost:6333")
|
||||
You will be using a pre-trained model called `sentence-transformers/all-MiniLM-L6-v2`.
|
||||
|
||||
```python
|
||||
qdrant_client.set_model("sentence-transformers/all-MiniLM-L6-v2")
|
||||
client.set_model("sentence-transformers/all-MiniLM-L6-v2")
|
||||
```
|
||||
|
||||
|
||||
4. Related vectors need to be added to a collection. Create a new collection for your startup vectors.
|
||||
|
||||
```python
|
||||
qdrant_client.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="startups",
|
||||
vectors_config=qdrant_client.get_fastembed_vector_params(),
|
||||
vectors_config=client.get_fastembed_vector_params(),
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -146,13 +146,13 @@ Now you need to write a script to upload all startup data and vectors into the s
|
||||
from qdrant_client import QdrantClient
|
||||
from qdrant_client.models import VectorParams, Distance
|
||||
|
||||
qdrant_client = QdrantClient("http://localhost:6333")
|
||||
client = QdrantClient("http://localhost:6333")
|
||||
```
|
||||
|
||||
3. Related vectors need to be added to a collection. Create a new collection for your startup vectors.
|
||||
|
||||
```python
|
||||
qdrant_client.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="startups",
|
||||
vectors_config=VectorParams(size=384, distance=Distance.COSINE),
|
||||
)
|
||||
@@ -186,7 +186,7 @@ vectors = np.load("./startup_vectors.npy")
|
||||
5. Upload the data
|
||||
|
||||
```python
|
||||
qdrant_client.upload_collection(
|
||||
client.upload_collection(
|
||||
collection_name="startups",
|
||||
vectors=vectors,
|
||||
payload=payload,
|
||||
|
||||
@@ -105,10 +105,10 @@ after receiving the response from the `upsert` endpoint. **As long as the indexi
|
||||
the exact search**. We have to wait until the indexing is finished to be sure that the approximate search is performed.
|
||||
|
||||
```python
|
||||
client.upload_records(
|
||||
client.upload_points( # upload_points is available as of qdrant-client v1.7.1
|
||||
collection_name="arxiv-titles-instructorxl-embeddings",
|
||||
records=[
|
||||
models.Record(
|
||||
points=[
|
||||
models.PointStruct(
|
||||
id=item["id"],
|
||||
vector=item["vector"],
|
||||
payload=item,
|
||||
|
||||
@@ -147,7 +147,7 @@ documents = [
|
||||
You need to tell Qdrant where to store embeddings. This is a basic demo, so your local computer will use its memory as temporary storage.
|
||||
|
||||
```python
|
||||
qdrant = QdrantClient(":memory:")
|
||||
client = QdrantClient(":memory:")
|
||||
```
|
||||
|
||||
## 4. Create a collection
|
||||
@@ -155,7 +155,7 @@ qdrant = QdrantClient(":memory:")
|
||||
All data in Qdrant is organized by collections. In this case, you are storing books, so we are calling it `my_books`.
|
||||
|
||||
```python
|
||||
qdrant.recreate_collection(
|
||||
client.recreate_collection(
|
||||
collection_name="my_books",
|
||||
vectors_config=models.VectorParams(
|
||||
size=encoder.get_sentence_embedding_dimension(), # Vector size is defined by used model
|
||||
@@ -176,7 +176,7 @@ qdrant.recreate_collection(
|
||||
Tell the database to upload `documents` to the `my_books` collection. This will give each record an id and a payload. The payload is just the metadata from the dataset.
|
||||
|
||||
```python
|
||||
qdrant.upload_points(
|
||||
client.upload_points(
|
||||
collection_name="my_books",
|
||||
points=[
|
||||
models.PointStruct(
|
||||
@@ -192,7 +192,7 @@ qdrant.upload_points(
|
||||
Now that the data is stored in Qdrant, you can ask it questions and receive semantically relevant results.
|
||||
|
||||
```python
|
||||
hits = qdrant.search(
|
||||
hits = client.search(
|
||||
collection_name="my_books",
|
||||
query_vector=encoder.encode("alien invasion").tolist(),
|
||||
limit=3,
|
||||
@@ -216,7 +216,7 @@ The search engine shows three of the most likely responses that have to do with
|
||||
How about the most recent book from the early 2000s?
|
||||
|
||||
```python
|
||||
hits = qdrant.search(
|
||||
hits = client.search(
|
||||
collection_name="my_books",
|
||||
query_vector=encoder.encode("alien invasion").tolist(),
|
||||
query_filter=models.Filter(
|
||||
|
||||
Reference in New Issue
Block a user