Merge branch 'master' into vector-database-revamp

This commit is contained in:
davidmyriel
2024-10-09 16:28:36 -07:00
168 changed files with 2735 additions and 1437 deletions
+2 -2
View File
@@ -27,11 +27,11 @@ jobs:
export PATH="${CURRENT_DIR}/dart-sass:${PATH}" export PATH="${CURRENT_DIR}/dart-sass:${PATH}"
cd qdrant-landing && hugo --gc -b 'http://localhost:1313' && hugo serve & cd qdrant-landing && hugo --gc -b 'http://localhost:1313' && hugo serve &
sleep 5 # wait for server to start sleep 5 # wait for server to start
- name: Link Checker - name: Internal Links Check
id: lychee id: lychee
uses: lycheeverse/lychee-action@v1.8.0 uses: lycheeverse/lychee-action@v1.8.0
with: with:
args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' qdrant-landing/public/ args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' --base http://localhost:1313/ qdrant-landing/public/
fail: true fail: true
env: env:
GITHUB_TOKEN: ${{secrets.GITHUB_TOKEN}} GITHUB_TOKEN: ${{secrets.GITHUB_TOKEN}}
-201
View File
@@ -101,207 +101,6 @@ disableKinds = ["taxonomy", "term"]
category = "categories" category = "categories"
example = "examples" example = "examples"
[menu]
[[menu.main]]
identifier = "product"
name = "Product"
weight = 1
[menu.main.params]
in_header = true
in_footer = true
[[menu.main]]
identifier = "use-case"
name = "Use cases"
weight = 1
parent = "product"
url = "/use-cases/"
[[menu.main]]
identifier = "solutions"
name = "Solutions"
weight = 2
parent = "product"
url = "/solutions/"
[[menu.main]]
identifier = "benchmarks"
name = "Benchmarks"
weight = 3
parent = "product"
url = "/benchmarks/"
[[menu.main]]
identifier = "demo"
name = "Demos"
weight = 4
parent = "product"
url = "/demo/"
[[menu.main]]
identifier = "pricing"
name = "Pricing"
weight = 5
parent = "product"
url = "/pricing/"
[[menu.main]]
identifier = "resources"
name = "Resources"
weight = 2
[menu.main.params]
in_header = true
in_footer = false
[[menu.main]]
identifier = "documentation"
name = "Documentation"
parent = "resources"
weight = 1
url = "/documentation/"
[[menu.main]]
identifier = "articles"
name = "Articles"
weight = 4
parent = "resources"
url = "/articles/"
[[menu.main]]
identifier = "blog"
name = "Blog"
weight = 5
parent = "resources"
url = "/blog/"
[[menu.main]]
identifier = "roadmap"
name = "Roadmap"
weight = 6
parent = "resources"
url = "https://qdrant.to/roadmap"
[menu.main.params]
external = true
[[menu.main]]
identifier = "changelog"
name = "Changelog"
weight = 7
parent = "resources"
url = "https://github.com/qdrant/qdrant/releases"
[menu.main.params]
external = true
[[menu.main]]
identifier = "trust-center"
name = "Trust Center"
weight = 8
parent = "resources"
url = "http://qdrant.to/trust-center"
[menu.main.params]
external = true
[[menu.main]]
identifier = "community"
name = "Community"
weight = 4
[menu.main.params]
in_header = true
in_footer = true
[[menu.main]]
identifier = "github"
name = "Github"
weight = 1
parent = "community"
url = "https://github.com/qdrant/qdrant"
pre = "<i class='fab fa-github'></i>"
[menu.main.params]
external = true
[[menu.main]]
identifier = "discord"
name = "Discord"
weight = 2
parent = "community"
url = "https://qdrant.to/discord"
pre = "<i class='fab fa-discord'></i>"
[menu.main.params]
external = true
[[menu.main]]
identifier = "twitter"
name = "Twitter"
weight = 3
parent = "community"
url = "https://qdrant.to/twitter"
pre = "<i class='fab fa-twitter'></i>"
[menu.main.params]
external = true
[[menu.main]]
identifier = "newsletter"
name = "Newsletter"
weight = 4
parent = "community"
url = "/subscribe/"
pre = "<i class='fas fa-mail-bulk'></i>"
[[menu.main]]
identifier = "contact"
name = "Contact us"
weight = 5
parent = "community"
url = "https://qdrant.to/contact-us"
pre = "<i class='fas fa-envelope'></i>"
[menu.main.params]
external = true
[[menu.main]]
identifier = "company"
name = "Company"
weight = 4
[menu.main.params]
in_header = false
in_footer = true
[[menu.main]]
identifier = "jobs"
name = "Jobs"
weight = 1
parent = "company"
url = "https://qdrant.join.com"
[[menu.main]]
identifier = "privacy-policy"
name = "Privacy Policy"
weight = 2
parent = "company"
url = "/legal/privacy-policy/"
[[menu.main]]
identifier = "terms"
name = "Terms"
weight = 3
parent = "company"
url = "/legal/terms_and_conditions/"
[[menu.main]]
identifier = "impressum"
name = "Impressum"
weight = 4
parent = "company"
url = "/legal/impressum/"
[[menu.main]]
identifier = "credits"
name = "Credits"
weight = 5
parent = "company"
url = "/legal/credits/"
[markup] [markup]
[markup.goldmark.renderer] [markup.goldmark.renderer]
unsafe=true unsafe=true
@@ -1,14 +1,13 @@
--- ---
draft: false title: "Mastering Batch Search for Vector Optimization"
title: Mastering Batch Search for Vector Optimization | Qdrant short_description: "Introducing efficient batch vector search capabilities, streamlining and optimizing large-scale searches for enhanced performance."
slug: batch-vector-search-with-qdrant
short_description: Introducing efficient batch vector search capabilities,
streamlining and optimizing large-scale searches for enhanced performance.
description: "Discover how to optimize your vector search capabilities with efficient batch search. Learn optimization strategies for faster, more accurate results." description: "Discover how to optimize your vector search capabilities with efficient batch search. Learn optimization strategies for faster, more accurate results."
preview_image: /blog/from_cms/andrey.vasnetsov_career_mining_on_the_moon_with_giant_machines_813bc56a-5767-4397-9243-217bea869820.png preview_dir: /articles_data/batch-vector-search-with-qdrant/preview
date: 2022-09-26T15:39:53.751Z social_preview_image: /articles_data/batch-vector-search-with-qdrant/preview/social_preview.jpg
author: Kacper Łukawski author: Kacper Łukawski
featured: false date: 2022-09-26T00:00:00-08:00
aliases:
- /blog/batch-vector-search-with-qdrant/
tags: tags:
- Data Science - Data Science
- Vector Database - Vector Database
@@ -0,0 +1,126 @@
---
title: Qdrant Summer of Code 2024 - WASM based Dimension Reduction
short_description: QSOC'24 WASM based Dimension Reduction
description: My journey as a Qdrant Summer of Code 2024 participant working on enhancing vector visualization using WebAssembly (WASM) based dimension reduction.
preview_dir: /articles_data/dimension-reduction-qsoc/preview
small_preview_image: /articles_data/dimension-reduction-qsoc/icon.svg
social_preview_image: /articles_data/dimension-reduction-qsoc/preview/social_preview.jpg
weight: -10
author: Jishan Bhattacharya
author_link: https://www.linkedin.com/in/j16n/
date: 2024-08-31T10:39:48.312Z
draft: false
keywords:
- dimension reduction
- web assembly
- qsoc'24
- vector similarity
- tsne
- qdrant data visualization
---
## Introduction
Hello, everyone! I'm Jishan Bhattacharya, and I had the incredible opportunity to intern at Qdrant this summer as part of the Qdrant Summer of Code 2024. Under the mentorship of [Andrey Vasnetsov](https://www.linkedin.com/in/andrey-vasnetsov-75268897/), I dived into the world of performance optimization, focusing on enhancing vector visualization using WebAssembly (WASM). In this article, I'll share the insights, challenges, and accomplishments from my journey — one filled with learning, experimentation, and plenty of coding adventures.
## Project Overview
Qdrant is a robust vector database and search engine designed to store vector data and perform tasks like similarity search and clustering. One of its standout features is the ability to visualize high-dimensional vectors in a 2D space. However, the existing implementation faced performance bottlenecks, especially when scaling to large datasets. My mission was to tackle this challenge by leveraging a WASM-based solution for dimensionality reduction in the visualization process.
## Learnings & Challenges
Our weapon of choice was Rust, paired with WASM, and we employed the t-SNE algorithm for dimensionality reduction. For those unfamiliar, t-SNE (t-Distributed Stochastic Neighbor Embedding) is a technique that helps visualize high-dimensional data by projecting it into two or three dimensions. It operates in two main steps:
1. **Computing Pairwise Similarity:** This step involves calculating the similarity between each pair of data points in the original high-dimensional space.
2. **Iterative Optimization:** The second step is iterative, where the embedding is refined using gradient descent. Here, the similarity matrix from the first step plays a crucial role.
At the outset, Andrey tasked me with rewriting the existing JavaScript implementation of t-SNE in Rust, introducing multi-threading along the way. Setting up WASM with Vite for multi-threaded execution was no small feat, but the effort paid off. The resulting Rust implementation outperformed the single-threaded JavaScript version, although it still struggled with large datasets.
Next came the challenge of optimizing the algorithm further. A key aspect of t-SNE's first step is finding the nearest neighbors for each data point, which requires an efficient data structure. I opted for a [Vantage Point Tree](https://en.wikipedia.org/wiki/Vantage-point_tree) (also known as a Ball Tree) to speed up this process. As for the second step, while it is inherently sequential, there was still room for improvement. I incorporated Barnes-Hut approximation to accelerate the gradient calculation. This method approximates the forces between points in low dimensional space, making the process more efficient.
To illustrate, imagine dividing a 2D space into quadrants, each containing multiple points. Every quadrant is again subdivided into four quadrants. This is done until every point belongs to a single cell.
{{< figure
src="/articles_data/dimension-reduction-qsoc/barnes_hut.png"
caption="Barnes-Hut Approximation"
alt="Calculating the resultant force on red point using Barnes-Hut approximation"
>}}
We then calculate the center of mass for each cell represented by a blue circle as shown in the figure. Now let’s say we want to find all the forces, represented by dotted lines, on the red point. Barnes Hut’s approximation states that for points that are sufficiently distant, instead of computing the force for each individual point, we use the center of mass as a proxy, significantly reducing the computational load. This is represented by the blue dotted line in the figure.
These optimizations made a remarkable difference — Barnes-Hut t-SNE was eight times faster than the exact t-SNE for 10,000 vectors.
{{< figure
src="/articles_data/dimension-reduction-qsoc/rust_rewrite.jpg"
caption="Exact t-SNE - Total time: 884.728s"
alt="Image of visualizing 10,000 vectors using exact t-SNE which took 884.728s"
>}}
{{< figure
src="/articles_data/dimension-reduction-qsoc/rust_bhtsne.jpg"
caption="Barnes-Hut t-SNE - Total time: 104.191s"
alt="Image of visualizing 10,000 vectors using Barnes-Hut t-SNE which took 110.728s"
>}}
Despite these improvements, the first step of the algorithm was still a bottleneck, leading to noticeable delays and blank screens. I experimented with approximate nearest neighbor algorithms, but the performance gains were minimal. After consulting with my mentor, we decided to compute the nearest neighbors on the server side, passing the distance matrix directly to the visualization process instead of the raw vectors.
While waiting for the distance-matrix API to be ready, I explored further optimizations. I observed that the worker thread sent results to the main thread for rendering at specific intervals, causing unnecessary delays due to serialization and deserialization.
{{< figure
src="/articles_data/dimension-reduction-qsoc/channels.png"
caption="Serialization and Deserialization Overhead"
alt="Image showing serialization and deserialization overhead due to message passing between threads"
>}}
To address this, I implemented a `SharedArrayBuffer`, allowing the main thread to access changes made by the worker thread instantly. This change led to noticeable improvements.
Additionally, the previous architecture resulted in choppy animations due to the fixed intervals at which the worker thread sent results.
{{< figure
src="/articles_data/dimension-reduction-qsoc/prev_arch.png"
caption="Previous architecture with fixed intervals"
alt="Image showing the previous architecture of the frontend with fixed intervals for sending results"
>}}
I introduced a "rendering-on-demand" approach, where the main thread would signal the worker thread when it was ready to render the next result. This created smoother, more responsive animations.
{{< figure
src="/articles_data/dimension-reduction-qsoc/curr_arch.png"
caption="Current architecture with rendering-on-demand"
alt="Image showing the current architecture of the frontend with rendering-on-demand approach"
>}}
With these optimizations in place, the final step was wrapping up the project by creating a Node.js [package](https://www.npmjs.com/package/wasm-dist-bhtsne). This package exposed the necessary interfaces to accept the distance matrix, perform calculations, and return the results, making the solution easy to integrate into various projects.
## Areas for Improvement
While reflecting on this transformative journey, there are still areas that offer room for improvement and future enhancements:
1. **Payload Parsing:** When requesting a large number of vectors, parsing the payload on the main thread can make the user interface unresponsive. Implementing a faster parser could mitigate this issue.
2. **Direct Data Requests:** Allowing the worker thread to request data directly could eliminate the initial transfer of data from the main thread, speeding up the overall process.
3. **Chart Library Optimization:** Profiling revealed that nearly 80% of the time was spent on the Chart.js update function. Switching to a WebGL-accelerated chart library could dramatically improve performance, especially for large datasets.
{{< figure
src="/articles_data/dimension-reduction-qsoc/profiling.png"
caption="Profiling Result"
alt="Image showing profiling results with 80% time spent on Chart.js update function"
>}}
## Conclusion
Participating in the Qdrant Summer of Code 2024 was a deeply rewarding experience. I had the chance to push the boundaries of my coding skills while exploring new technologies like Rust and WebAssembly. I'm incredibly grateful for the guidance and support from my mentor and the entire Qdrant team, who made this journey both educational and enjoyable.
This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide.
This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide.
Thank you for joining me on this coding adventure. I hope you found something valuable in my journey, and I look forward to sharing more exciting projects with you in the future. Happy coding!
@@ -1,20 +1,21 @@
--- ---
draft: false title: "Full-text filter and index are already available!"
title: Full-text filter and index are already available!
slug: qdrant-introduces-full-text-filters-and-indexes slug: qdrant-introduces-full-text-filters-and-indexes
short_description: Qdrant v0.10 introduced full-text filters short_description: "Qdrant v0.10 introduced full-text filters."
description: Qdrant v0.10 introduced full-text filters and indexes to enable description: "Qdrant v0.10 introduced full-text filters and indexes to enable more search capabilities for those working with textual data."
more search capabilities for those working with textual data. preview_dir: /articles_data/qdrant-introduces-full-text-filters-and-indexes/preview
preview_image: /blog/from_cms/andrey.vasnetsov_black_hole_sucking_up_the_word_tag_cloud_f349586d-3e51-43c5-9e5e-92abf9a9e871.png social_preview_image: /articles_data/qdrant-introduces-full-text-filters-and-indexes/preview/social_preview.jpg
date: 2022-11-16T09:53:05.860Z
author: Kacper Łukawski author: Kacper Łukawski
featured: false date: 2022-11-16T00:00:00-08:00
aliases:
- /blog/qdrant-introduces-full-text-filters-and-indexes/
tags: tags:
- Information Retrieval - Information Retrieval
- Database - Database
- Open Source - Open Source
- Vector Search Database - Vector Search Database
--- ---
Qdrant is designed as an efficient vector database, allowing for a quick search of the nearest neighbours. But, you may find yourself in need of applying some extra filtering on top of the semantic search. Up to version 0.10, Qdrant was offering support for keywords only. Since 0.10, there is a possibility to apply full-text constraints as well. There is a new type of filter that you can use to do that, also combined with every other filter type. Qdrant is designed as an efficient vector database, allowing for a quick search of the nearest neighbours. But, you may find yourself in need of applying some extra filtering on top of the semantic search. Up to version 0.10, Qdrant was offering support for keywords only. Since 0.10, there is a possibility to apply full-text constraints as well. There is a new type of filter that you can use to do that, also combined with every other filter type.
## Using full-text filters without the payload index ## Using full-text filters without the payload index
@@ -51,7 +51,7 @@ Things have changed since then, as so many of you wanted a single tool for spars
If you're coming across the topic of sparse vectors for the first time, our [Brief History of Search](/documentation/overview/vector-search/) explains the difference between sparse and dense vectors. If you're coming across the topic of sparse vectors for the first time, our [Brief History of Search](/documentation/overview/vector-search/) explains the difference between sparse and dense vectors.
Check out the [sparse vectors article](../sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users. Check out the [sparse vectors article](/articles/sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users.
### Discovery API ### Discovery API
@@ -1,14 +1,14 @@
--- ---
title: "Semantic Cache: Accelerating AI with Lightning-Fast Data Retrieval" title: "Semantic Cache: Accelerating AI with Lightning-Fast Data Retrieval"
draft: false
slug:
short_description: "Semantic Cache for Best Results and Optimization." short_description: "Semantic Cache for Best Results and Optimization."
description: "Semantic cache is reshaping AI applications by enabling rapid data retrieval. Discover how its implementation benefits your RAG setup." description: "Semantic cache is reshaping AI applications by enabling rapid data retrieval. Discover how its implementation benefits your RAG setup."
preview_image: /blog/semantic-cache-ai-data-retrieval/social_preview.png preview_dir: /articles_data/semantic-cache-ai-data-retrieval/preview
social_preview_image: /blog/semantic-cache-ai-data-retrieval/social_preview.png social_preview_image: /articles_data/semantic-cache-ai-data-retrieval/preview/social_preview.jpg
date: 2024-05-07T00:00:00-08:00
author: Daniel Romero, David Myriel author: Daniel Romero, David Myriel
featured: false author_link: https://github.com/davidmyriel
date: 2024-05-07T00:00:00-08:00
aliases:
- /blog/semantic-cache-ai-data-retrieval/
tags: tags:
- vector search - vector search
- vector database - vector database
@@ -19,6 +19,7 @@ tags:
- data retrieval - data retrieval
- efficient data storage - efficient data storage
--- ---
## What is Semantic Cache? ## What is Semantic Cache?
**Semantic cache** is a method of retrieval optimization, where similar queries instantly retrieve the same appropriate response from a knowledge base. **Semantic cache** is a method of retrieval optimization, where similar queries instantly retrieve the same appropriate response from a knowledge base.
@@ -27,7 +28,7 @@ Semantic cache differs from traditional caching methods. In computing, **cache**
> The term **"semantic"** implies that the cache takes into account the meaning or semantics of the data or computation being cached, rather than just its syntactic representation. This can lead to more efficient caching strategies that exploit the structure or relationships within the data or computation. > The term **"semantic"** implies that the cache takes into account the meaning or semantics of the data or computation being cached, rather than just its syntactic representation. This can lead to more efficient caching strategies that exploit the structure or relationships within the data or computation.
![semantic-cache-question](/blog/semantic-cache-ai-data-retrieval/semantic-cache-question.png) ![semantic-cache-question](/articles_data/semantic-cache-ai-data-retrieval/semantic-cache-question.png)
Traditional caches operate on an exact match basis, while semantic caches search for the meaning of the key rather than an exact match. For example, **"What is the capital of Brazil?"** and **"Can you tell me the capital of Brazil?"** are semantically equivalent, but not exact matches. A semantic cache recognizes such semantic equivalence and provides the correct result. Traditional caches operate on an exact match basis, while semantic caches search for the meaning of the key rather than an exact match. For example, **"What is the capital of Brazil?"** and **"Can you tell me the capital of Brazil?"** are semantically equivalent, but not exact matches. A semantic cache recognizes such semantic equivalence and provides the correct result.
@@ -43,7 +44,7 @@ Qdrant is recommended for setting up semantic cache as semantically [evaluates](
**Diagram:** Semantic cache improves [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers. **Diagram:** Semantic cache improves [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers.
![Alt Text](/blog/semantic-cache-ai-data-retrieval/semantic-cache.gif) ![Alt Text](/articles_data/semantic-cache-ai-data-retrieval/semantic-cache.gif)
When using a key-value cache, it's important to consider that slight variations in question wording can lead to different hash values. The two questions convey the same query but differ in wording. A naive cache search might fail due to distinct hashed versions of the questions. Implementing a more nuanced approach is necessary to accommodate phrasing variations and ensure accurate responses. When using a key-value cache, it's important to consider that slight variations in question wording can lead to different hash values. The two questions convey the same query but differ in wording. A naive cache search might fail due to distinct hashed versions of the questions. Implementing a more nuanced approach is necessary to accommodate phrasing variations and ensure accurate responses.
@@ -1,14 +1,13 @@
--- ---
draft: false title: "Optimizing Semantic Search by Managing Multiple Vectors"
title: Optimizing Semantic Search by Managing Multiple Vectors short_description: "Qdrant's approach to storing multiple vectors per object, unraveling new possibilities in data representation and retrieval."
slug: storing-multiple-vectors-per-object-in-qdrant description: "Discover the power of vector storage optimization and learn how to efficiently manage multiple vectors per object for enhanced semantic search capabilities."
short_description: Qdrant's approach to storing multiple vectors per object, preview_dir: /articles_data/storing-multiple-vectors-per-object-in-qdrant/preview
unraveling new possibilities in data representation and retrieval. social_preview_image: /articles_data/storing-multiple-vectors-per-object-in-qdrant/preview/social_preview.jpg
description: Discover the power of vector storage optimization and learn how to efficiently manage multiple vectors per object for enhanced semantic search capabilities.
preview_image: /blog/from_cms/andrey.vasnetsov_a_space_station_with_multiple_attached_modules_853a27c7-05c4-45d2-aebc-700a6d1e79d0.png
date: 2022-10-05T10:05:43.329Z
author: Kacper Łukawski author: Kacper Łukawski
featured: false date: 2022-10-05T00:00:00-08:00
aliases:
- /blog/storing-multiple-vectors-per-object-in-qdrant/
tags: tags:
- Data Science - Data Science
- Neural Networks - Neural Networks
@@ -2,12 +2,12 @@
title: "What is Vector Quantization?" title: "What is Vector Quantization?"
draft: false draft: false
slug: what-is-vector-quantization slug: what-is-vector-quantization
short_description: What is Vector Quantization? Methods & Examples | Qdrant short_description: What is Vector Quantization? Methods & Examples
description: Learn what vector quantization is and explore how methods like Scalar, Product, and Binary Quantization work. Plus, find out how to choose the best method for your specific application. description: In this article, we'll teach you about compression methods like Scalar, Product, and Binary Quantization. Learn how to choose the best method for your specific application.
preview_dir: /articles_data/what-is-vector-quantization/preview preview_dir: /articles_data/what-is-vector-quantization/preview
weight: -210 weight: -210
social_preview_image: /articles_data/what-is-vector-quantization/preview/social-preview.jpg social_preview_image: /articles_data/what-is-vector-quantization/preview/social-preview.jpg
date: 2024-09-16T09:29:33-03:00 date: 2024-09-25T09:29:33-03:00
author: Sabrina Aquino author: Sabrina Aquino
featured: true featured: true
tags: tags:
@@ -20,33 +20,29 @@ tags:
--- ---
> Vector quantization is a technique for reducing the size of high-dimensional data. Instead of storing the original vectors, the data is stored in a compressed format, which reduces memory usage while maintaining most of the essential information. This compression allows for more efficient storage and faster search operations, especially in large datasets. Vector quantization is a data compression technique used to reduce the size of high-dimensional data. Compressing vectors reduces memory usage while maintaining nearly all of the essential information. This method allows for more efficient storage and faster search operations, particularly in large datasets.
When working with high-dimensional vectors, such as embeddings from AI models like OpenAI, one single 1536-dimensional vector requires **6KB of memory**. When working with high-dimensional vectors, such as embeddings from providers like OpenAI, a single 1536-dimensional vector requires **6 KB of memory**.
<img src="/articles_data/what-is-vector-quantization/vector-size.png" alt="1536 dimentional vector size is 6KB" width="700"> <img src="/articles_data/what-is-vector-quantization/vector-size.png" alt="1536-dimensional vector size is 6 KB" width="700">
So with 1 million vectors needing around 6GB of memory, as your dataset grows to multiple **millions of vectors**, the memory and processing demands increase significantly. With 1 million vectors needing around 6 GB of memory, as your dataset grows to multiple **millions of vectors**, the memory and processing demands increase significantly.
To understand why this process is so computational demanding, let's take a look at the nature of the [HNSW index.](https://qdrant.tech/documentation/concepts/indexing/#vector-index)
To understand why this process is so computationally demanding, let's take a look at the nature of the [HNSW index](https://qdrant.tech/documentation/concepts/indexing/#vector-index).
The **HNSW (Hierarchical Navigable Small World) index** organizes vectors in a layered graph, connecting each vector to its nearest neighbors. At each layer, the algorithm narrows down the search area until it reaches the lower layers, where it efficiently finds the closest matches to the query. The **HNSW (Hierarchical Navigable Small World) index** organizes vectors in a layered graph, connecting each vector to its nearest neighbors. At each layer, the algorithm narrows down the search area until it reaches the lower layers, where it efficiently finds the closest matches to the query.
<img src="/articles_data/what-is-vector-quantization/hnsw.png" alt="HNSW Search visualization" width="500"> <img src="/articles_data/what-is-vector-quantization/hnsw.png" alt="HNSW Search visualization" width="500">
So each time a new vector is added, the system must determine its position in the existing graph, a process similar to searching. This makes both inserting and searching for vectors complex operations. Each time a new vector is added, the system must determine its position in the existing graph, a process similar to searching. This makes both inserting and searching for vectors complex operations.
One of the key challenges with the HNSW index is that it requires a lot of **random reads** and **sequential traversals** through the graph. This makes the process computationally expensive, especially when you're dealing with millions of high-dimensional vectors. One of the key challenges with the HNSW index is that it requires a lot of **random reads** and **sequential traversals** through the graph. This makes the process computationally expensive, especially when you're dealing with millions of high-dimensional vectors.
The system has to jump between various points in the graph in an unpredictable way. This unpredictability makes it hard to optimize, and as the dataset grows, the memory and processing requirements increase significantly. The system has to jump between various points in the graph in an unpredictable manner. This unpredictability makes optimization difficult, and as the dataset grows, the memory and processing requirements increase significantly.
<img src="/articles_data/what-is-vector-quantization/hnsw-search2.png" alt="HNSW Search visualization" width="600"> <img src="/articles_data/what-is-vector-quantization/hnsw-search2.png" alt="HNSW Search visualization" width="600">
Since vectors need to be stored in **fast storage** like **RAM** or **SSD** for low-latency searches, as the size of the data grows, so does the cost of storing and processing it efficiently.
And because vectors need to be stored in **fast storage** like **RAM** or **SSD** for low-latency searches, as the size of the data grows, so does the cost of storing and processing it efficiently.
**Quantization** offers a solution by compressing vectors to smaller memory sizes, making the process more efficient. **Quantization** offers a solution by compressing vectors to smaller memory sizes, making the process more efficient.
@@ -54,14 +50,13 @@ There are several methods to achieve this, and here we will focus on three main
<img src="/articles_data/what-is-vector-quantization/types-of-quant.png" alt="Types of Quantization: 1. Scalar Quantization, 2. Product Quantization, 3. Binary Quantization" width="700"> <img src="/articles_data/what-is-vector-quantization/types-of-quant.png" alt="Types of Quantization: 1. Scalar Quantization, 2. Product Quantization, 3. Binary Quantization" width="700">
## 1. What is Scalar Quantization? ## 1. What is Scalar Quantization?
![](/articles_data/what-is-vector-quantization/astronaut-mars.jpg) ![](/articles_data/what-is-vector-quantization/astronaut-mars.jpg)
In Qdrant, each dimension is represented by a `float32` value, which uses **4 bytes** of memory. When using [Scalar Quantization](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization), we are mapping our vectors to a range that the smaller `int8` type can represent. An `int8` is only **1 byte** and can represent 256 values (from -128 to 127, or 0 to 255). This results in a **75% reduction** in memory size. In Qdrant, each dimension is represented by a `float32` value, which uses **4 bytes** of memory. When using [Scalar Quantization](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization), we map our vectors to a range that the smaller `int8` type can represent. An `int8` is only **1 byte** and can represent 256 values (from -128 to 127, or 0 to 255). This results in a **75% reduction** in memory size.
For example, if our data lies in the identified range of -1.0 to 1.0, Scalar Quantization will transform these values to a range that `int8` can represent, that is, within -128 to 127. So, the system **maps** the `float32` values into this range. For example, if our data lies in the range of -1.0 to 1.0, Scalar Quantization will transform these values to a range that `int8` can represent, i.e., within -128 to 127. The system **maps** the `float32` values into this range.
Here's a simple linear example of what this process looks like: Here's a simple linear example of what this process looks like:
@@ -100,24 +95,24 @@ client.create_collection(
) )
``` ```
The `quantile` is used to calculate the quantization bounds. For example, if you specify `0.99` as the quantile, 1% of extreme values will be excluded from the quantization bounds. The `quantile` parameter is used to calculate the quantization bounds. For example, if you specify a `0.99` quantile, the most extreme 1% of values will be excluded from the quantization bounds.
This parameter only affects the resulting precision, not the memory footprint. You can tune it if you experience a significant decrease in search quality. This parameter only affects the resulting precision, not the memory footprint. You can adjust it if you experience a significant decrease in search quality.
Scalar Quantization is a great choice if you're looking to boost search speed and compression without losing much accuracy. It also slighly improves performance, as distance calculations (such as dot product or cosine similarity) using `int8` values are computationally simpler than using `float32` values. Scalar Quantization is a great choice if you're looking to boost search speed and compression without losing much accuracy. It also slightly improves performance, as distance calculations (such as dot product or cosine similarity) using `int8` values are computationally simpler than using `float32` values.
While the performance gains of Scalar Quantization may not reach the levels seen with Binary Quantization (which we'll discuss later), it remains an excellent default choice when Binary Quantization isn’t the right fit for your use case. While the performance gains of Scalar Quantization may not match those achieved with Binary Quantization (which we'll discuss later), it remains an excellent default choice when Binary Quantization isn’t suitable for your use case.
# 2. What is Binary Quantization? ## 2. What is Binary Quantization?
![](/articles_data/what-is-vector-quantization/astronaut-white-surreal.jpg) ![Astronaut in surreal white environment](/articles_data/what-is-vector-quantization/astronaut-white-surreal.jpg)
[Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) is an excellent option if you're looking to **reduce memory** usage while also achieving a significant **boost in speed**. It works by converting high-dimensional vectors into simple binary (0 or 1) representations. [Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) is an excellent option if you're looking to **reduce memory** usage while also achieving a significant **boost in speed**. It works by converting high-dimensional vectors into simple binary (0 or 1) representations.
* Values greater than zero are converted to 1 - Values greater than zero are converted to 1.
* Values less than or equal to zero are converted to 0 - Values less than or equal to zero are converted to 0.
Let's take our initial example of a 1536-dimensional vector that requires **6KB** of memory (4 bytes for each `float32` value). Let's consider our initial example of a 1536-dimensional vector that requires **6 KB** of memory (4 bytes for each `float32` value).
After Binary Quantization, each dimension is reduced to 1 bit (1/8 byte), so the memory required is: After Binary Quantization, each dimension is reduced to 1 bit (1/8 byte), so the memory required is:
@@ -125,17 +120,14 @@ $$
\frac{1536 \text{ dimensions}}{8 \text{ bits per byte}} = 192 \text{ bytes} \frac{1536 \text{ dimensions}}{8 \text{ bits per byte}} = 192 \text{ bytes}
$$ $$
This leads to a **32x** memory reduction.
This leads to a **32x** memory saving.
<img src="/articles_data/what-is-vector-quantization/binary-quant.png" alt="Binary Quantization example" width="800"> <img src="/articles_data/what-is-vector-quantization/binary-quant.png" alt="Binary Quantization example" width="800">
Qdrant automates the Binary Quantization process during indexing. As vectors are added to your collection, each 32-bit floating-point component is converted into a binary value according to the configuration you define. Qdrant automates the Binary Quantization process during indexing. As vectors are added to your collection, each 32-bit floating-point component is converted into a binary value according to the configuration you define.
Here’s how you can set it up: Here’s how you can set it up:
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
{ {
@@ -163,7 +155,7 @@ client.create_collection(
) )
``` ```
Binary Quantization is by far the quantization method that will give you the most processing **speed gains** when compared to Scalar and Product Quantizations. This is because the binary representation allows the system to use highly optimized CPU instructions, such as [XOR](https://en.wikipedia.org/wiki/XOR_gate#:~:text=XOR%20represents%20the%20inequality%20function,the%20other%20but%20not%20both%22.) and [Popcount](https://en.wikipedia.org/wiki/Hamming_weight), for fast distance computations. Binary Quantization is by far the quantization method that provides the most significant processing **speed gains** compared to Scalar and Product Quantizations. This is because the binary representation allows the system to use highly optimized CPU instructions, such as [XOR](https://en.wikipedia.org/wiki/XOR_gate#:~:text=XOR%20represents%20the%20inequality%20function,the%20other%20but%20not%20both%22) and [Popcount](https://en.wikipedia.org/wiki/Hamming_weight), for fast distance computations.
It can speed up search operations by **up to 40x**, depending on the dataset and hardware. It can speed up search operations by **up to 40x**, depending on the dataset and hardware.
@@ -171,29 +163,28 @@ Not all models are equally compatible with Binary Quantization, and in the compa
The models that have shown the best compatibility with this method include: The models that have shown the best compatibility with this method include:
* **OpenAI text-embedding-ada-002** (1536 dimensions) - **OpenAI text-embedding-ada-002** (1536 dimensions)
* **Cohere AI embed-english-v2.0** (4096 dimensions) - **Cohere AI embed-english-v2.0** (4096 dimensions)
They demonstrate minimal accuracy loss while still benefiting from the substantial speed and memory gains. These models demonstrate minimal accuracy loss while still benefiting from substantial speed and memory gains.
Even though Binary Quantization is incredibly fast and memory-efficient, the trade-offs are in **precision** and **model compatibility**, and you may need to ensure search quality using techniques like oversampling and rescoring. Even though Binary Quantization is incredibly fast and memory-efficient, the trade-offs are in **precision** and **model compatibility**, so you may need to ensure search quality using techniques like oversampling and rescoring.
If you're interested in exploring Binary Quantization in more detail—including implementation examples, benchmark results, and usage recommendations—check out our dedicated article on [Binary Quantization - Vector Search, 40x Faster](https://qdrant.tech/articles/binary-quantization/). If you're interested in exploring Binary Quantization in more detail—including implementation examples, benchmark results, and usage recommendations—check out our dedicated article on [Binary Quantization - Vector Search, 40x Faster](https://qdrant.tech/articles/binary-quantization/).
# 3. What is Product Quantization? ## 3. What is Product Quantization?
![](/articles_data/what-is-vector-quantization/astronaut-centroids.jpg) ![](/articles_data/what-is-vector-quantization/astronaut-centroids.jpg)
[Product Quantization](https://qdrant.tech/documentation/guides/quantization/#product-quantization) is a method used to compress high-dimensional vectors by representing them with a smaller set of representative points. [Product Quantization](https://qdrant.tech/documentation/guides/quantization/#product-quantization) is a method used to compress high-dimensional vectors by representing them with a smaller set of representative points.
The process begins by splitting the original high-dimensional vectors into smaller **sub-vectors.** Each sub-vector represents a segment of the original vector, which can capture different characteristics of the data. The process begins by splitting the original high-dimensional vectors into smaller **sub-vectors.** Each sub-vector represents a segment of the original vector, capturing different characteristics of the data.
<img src="/articles_data/what-is-vector-quantization/subvec.png" alt="Creation of the Suba-vector" width="700"> <img src="/articles_data/what-is-vector-quantization/subvec.png" alt="Creation of the Sub-vector" width="700">
For each sub-vector, a separate **codebook** is created, representing regions in the data space where common patterns occur. For each sub-vector, a separate **codebook** is created, representing regions in the data space where common patterns occur.
The codebook in Qdrant is trained automatically during the indexation process. As vectors are added to the collection, Qdrant uses your specified quantization settings in the `quantization_config` to build the codebook and quantize the vectors. Here’s how you might set it up: The codebook in Qdrant is trained automatically during the indexing process. As vectors are added to the collection, Qdrant uses your specified quantization settings in the `quantization_config` to build the codebook and quantize the vectors. Here’s how you can set it up:
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -225,25 +216,23 @@ client.create_collection(
) )
``` ```
Each region in the codebook is defined by a **centroid**, which serves as a representative point that summarizes the characteristics of that region. So, instead of treating every single data point as equally important, we can group similar sub-vectors together and represent them with a single centroid that captures the general characteristics of that group. Each region in the codebook is defined by a **centroid**, which serves as a representative point summarizing the characteristics of that region. Instead of treating every single data point as equally important, we can group similar sub-vectors together and represent them with a single centroid that captures the general characteristics of that group.
The centroids used in Product Quantization are determined using the **[K-means clustering algorithm](https://en.wikipedia.org/wiki/K-means_clustering)**. The centroids used in Product Quantization are determined using the **[K-means clustering algorithm](https://en.wikipedia.org/wiki/K-means_clustering)**.
<img src="/articles_data/what-is-vector-quantization/code-book.png" alt="Codebook and Centroids example" width="700"> <img src="/articles_data/what-is-vector-quantization/code-book.png" alt="Codebook and Centroids example" width="700">
Qdrant always selects **K = 256** as the number of centroids in its implementation, based on the fact that 256 is the maximum number of unique values that can be represented by a single byte.
Qdrant always selects **K = 256** for the number of centroids in its implementation based on the fact that 256 is the maximum number of unique values that can be represented by a single byte.
This makes the compression process efficient because each centroid index can be stored in a single byte. This makes the compression process efficient because each centroid index can be stored in a single byte.
The original high-dimensional vectors are quantized by mapping each sub-vector to the nearest centroid in its respective codebook. The original high-dimensional vectors are quantized by mapping each sub-vector to the nearest centroid in its respective codebook.
<img src="/articles_data/what-is-vector-quantization/mapping.png" alt="Vectors being mapped to their correspondent centroids example" width="700"> <img src="/articles_data/what-is-vector-quantization/mapping.png" alt="Vectors being mapped to their corresponding centroids example" width="700">
The compressed vector stores the index of the closest centroid for each sub-vector. The compressed vector stores the index of the closest centroid for each sub-vector.
Here’s how a 1024-dimensional vector originally taking up 4096 bytes is reduced to just 128 bytes by representing it as 128 indexes, each pointing to the centroid of a sub-vector: Here’s how a 1024-dimensional vector, originally taking up 4096 bytes, is reduced to just 128 bytes by representing it as 128 indexes, each pointing to the centroid of a sub-vector:
<img src="/articles_data/what-is-vector-quantization/product-quant.png" alt="Product Quantization example" width="800"> <img src="/articles_data/what-is-vector-quantization/product-quant.png" alt="Product Quantization example" width="800">
@@ -275,12 +264,11 @@ client.query_points(
limit=10 # Return the top 10 results limit=10 # Return the top 10 results
) )
``` ```
Product Quantization can significantly reduce memory usage, potentially offering up to **64x** compression in certain configurations. However, it's important to note that this level of compression can lead to a noticeable drop in quality. Product Quantization can significantly reduce memory usage, potentially offering up to **64x** compression in certain configurations. However, it's important to note that this level of compression can lead to a noticeable drop in quality.
If your application needs high precision or real-time performance, Product Quantization may not be the best choice. But if **memory savings** are critical and some accuracy loss is acceptable, it could still be the ideal solution. If your application requires high precision or real-time performance, Product Quantization may not be the best choice. However, if **memory savings** are critical and some accuracy loss is acceptable, it could still be an ideal solution.
Here's the speed, accuracy, and compressio comparison of all three methods, adapted from [Qdrant's documentation](https://qdrant.tech/documentation/guides/quantization/#how-to-choose-the-right-quantization-method): Here’s a comparison of speed, accuracy, and compression for all three methods, adapted from [Qdrant's documentation](https://qdrant.tech/documentation/guides/quantization/#how-to-choose-the-right-quantization-method):
| Quantization method | Accuracy | Speed | Compression | | Quantization method | Accuracy | Speed | Compression |
|---------------------|----------|------------|-------------| |---------------------|----------|------------|-------------|
@@ -290,17 +278,17 @@ Here's the speed, accuracy, and compressio comparison of all three methods, adap
\* - for compatible models \* - for compatible models
For a more in-depth understanting of the benchmarks you can expect, check out our dedicated article on [Product Quantization in Vector Search](https://qdrant.tech/articles/product-quantization/). For a more in-depth understanding of the benchmarks you can expect, check out our dedicated article on [Product Quantization in Vector Search](https://qdrant.tech/articles/product-quantization/).
# Understanding Rescoring, Oversampling, and Reranking ## Rescoring, Oversampling, and Reranking
When we use quantization methods like Scalar, Binary, or Product Quantization, we're compressing our vectors to save memory and improve performance. However, this compression strips away some detail from the original vectors. When we use quantization methods like Scalar, Binary, or Product Quantization, we're compressing our vectors to save memory and improve performance. However, this compression removes some detail from the original vectors.
This can slightly reduce the accuracy of our similarity searches because the quantized vectors are approximations of the original data. To mitigate this loss of accuracy, you can use **oversampling** and **rescoring**, which help improve the accuracy of the final search results. This can slightly reduce the accuracy of our similarity searches because the quantized vectors are approximations of the original data. To mitigate this loss of accuracy, you can use **oversampling** and **rescoring**, which help improve the accuracy of the final search results.
The original vectors are never deleted during this process, and you can easily switch between quantization methods or parameters by updating the collection method at any time. The original vectors are never deleted during this process, and you can easily switch between quantization methods or parameters by updating the collection configuration at any time.
Here's how the process works, step by step: Here’s how the process works, step by step:
### 1. Initial Quantized Search ### 1. Initial Quantized Search
@@ -310,37 +298,36 @@ When you perform a search, Qdrant retrieves the top candidates using the quantiz
### 2. Oversampling ### 2. Oversampling
Oversampling is a technique that helps make up for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results. Oversampling is a technique that helps compensate for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results.
You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results (`limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2). You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results (`limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2).
<img src="/articles_data/what-is-vector-quantization/ann-search-quantized-oversampling.png" alt="ANN Search with Quantization and Oversampling" width="600"> <img src="/articles_data/what-is-vector-quantization/ann-search-quantized-oversampling.png" alt="ANN Search with Quantization and Oversampling" width="600">
You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of getting high-quality top-K results, especially after rescoring with original vectors. You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of obtaining high-quality top-K results, especially after rescoring with the original vectors.
### 3. Rescoring with Original Vectors ### 3. Rescoring with Original Vectors
After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query. After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query.
The rescoring process **maps** the quantized vectors to the corresponding original vectors and allows you to consider factors like context, metadata, or additional relevance that wasn't included in the initial search, leading to more accurate results. The rescoring process **maps** the quantized vectors to their corresponding original vectors, allowing you to consider factors like context, metadata, or additional relevance that wasn’t included in the initial search, leading to more accurate results.
![Rescoring with Original Vectors](/articles_data/what-is-vector-quantization/rescoring.png) ![Rescoring with Original Vectors](/articles_data/what-is-vector-quantization/rescoring.png)
During rescoring, one of the lower-ranked candidates from oversampling might turn out to be a better match than some of the original top-K candidates. During rescoring, one of the lower-ranked candidates from oversampling might turn out to be a better match than some of the original top-K candidates.
Even though rescoring uses the original, larger vectors, the process remains much faster because only a very small number of vectors are read. Since the initial quantized search already identifies the specific vectors to read, rescore, and rerank. Even though rescoring uses the original, larger vectors, the process remains much faster because only a very small number of vectors are read. The initial quantized search already identifies the specific vectors to read, rescore, and rerank.
### 4. Reranking ### 4. Reranking
With the new similarity scores from rescoring, **reranking** is where the final top-K candidates are determined based on the updated similarity scores. With the new similarity scores from rescoring, **reranking** is where the final top-K candidates are determined based on the updated similarity scores.
For example, in our case with a limit of 4, one candidate that ranked 6th in the quantized search might improve its score after rescoring because the original vectors capture more context or metadata. As a result, this candidate could move into the final top 4 after reranking, replacing a less relevant option from the initial search. For example, in our case with a limit of 4, a candidate that ranked 6th in the initial quantized search might improve its score after rescoring because the original vectors capture more context or metadata. As a result, this candidate could move into the final top 4 after reranking, replacing a less relevant option from the initial search.
<img src="/articles_data/what-is-vector-quantization/reranking.png" alt="Reranking with Original Vectors" width="600"> <img src="/articles_data/what-is-vector-quantization/reranking.png" alt="Reranking with Original Vectors" width="600">
Here's how you can set it up: Here's how you can set it up:
```http ```http
POST /collections/{collection_name}/points/search POST /collections/{collection_name}/points/search
@@ -373,12 +360,11 @@ client.query_points(
You can adjust the `oversampling` factor to find the right balance between search speed and result accuracy. You can adjust the `oversampling` factor to find the right balance between search speed and result accuracy.
If quantization is affecting performance in an application that needs high accuracy, combining oversampling with rescoring is a great choice. But if you need faster searches and can tolerate some loss in accuracy, you might choose to use oversampling without rescoring, or adjust the oversampling factor to a lower value. If quantization is impacting performance in an application that requires high accuracy, combining oversampling with rescoring is a great choice. However, if you need faster searches and can tolerate some loss in accuracy, you might choose to use oversampling without rescoring, or adjust the oversampling factor to a lower value.
## Distributing Resources Between Disk & Memory
### Disk & RAM Storage Tuning Qdrant stores both the quantized and original vectors. When you enable quantization, both the original and quantized vectors are stored in RAM by default. You can move the original vectors to disk to significantly reduce RAM usage and lower system costs. Simply enabling quantization is not enough—you need to explicitly move the original vectors to disk by setting `on_disk=True`.
Qdrant stores both the quantized and original vectors. When you enable quantization, both the original and quantized vectors are stored in RAM by default. You can leave the original vectors on disk to significantly reduce RAM usage and reduce system costs. Just enabling quantization won't be enough—you need to explicitly move the original vectors to disk by setting `on_disk=True`.
Here’s an example configuration: Here’s an example configuration:
@@ -414,32 +400,30 @@ client.update_collection(
) )
``` ```
Without explicitly setting `on_disk=True`, you won't see any RAM savings, even with quantization enabled. So, ensure that you configure both storage and quantization options based on your memory and performance needs. If your storage has high disk latency, you can try disabling rescoring to maintain speed. Without explicitly setting `on_disk=True`, you won't see any RAM savings, even with quantization enabled. So, make sure to configure both storage and quantization options based on your memory and performance needs. If your storage has high disk latency, consider disabling rescoring to maintain speed.
#### Speeding Up Rescoring with io_uring ### Speeding Up Rescoring with io_uring
When dealing with large collections of quantized vectors, frequent disk reads are required to retrieve both original and compressed data for rescoring operations. While `mmap` helps with efficient I/O by reducing user-to-kernel transitions, rescoring can still be slowed down when working with large datasets on disk because frequent disk reads are needed. When dealing with large collections of quantized vectors, frequent disk reads are required to retrieve both original and compressed data for rescoring operations. While `mmap` helps with efficient I/O by reducing user-to-kernel transitions, rescoring can still be slowed down when working with large datasets on disk due to the need for frequent disk reads.
On Linux-based systems, `io_uring` allows multiple disk operations to be processed in parallel, significantly reducing I/O overhead. This optimization is particularly effective during rescoring, where multiple vectors need to be re-evaluated after an initial search. With io_uring, Qdrant can retrieve and rescore vectors from disk in the most efficient way, improving overall search efficiency. On Linux-based systems, `io_uring` allows multiple disk operations to be processed in parallel, significantly reducing I/O overhead. This optimization is particularly effective during rescoring, where multiple vectors need to be re-evaluated after the initial search. With io_uring, Qdrant can retrieve and rescore vectors from disk in the most efficient way, improving overall search performance.
When you perform vector quantization and store data on disk, Qdrant often needs to access multiple vectors in parallel. Without io_uring, this process can slow down because of the system’s limitations in handling many disk accesses. When you perform vector quantization and store data on disk, Qdrant often needs to access multiple vectors in parallel. Without io_uring, this process can be slowed down due to the system’s limitations in handling many disk accesses.
To enable `io_uring` in Qdrant, add the following to your storage configuration: To enable `io_uring` in Qdrant, add the following to your storage configuration:
```http
```yaml
storage: storage:
async_scorer: true # Enable io_uring for async storage async_scorer: true # Enable io_uring for async storage
``` ```
Without this configuration, Qdrant will default to using mmap for disk I/O operations.
Without this configuration, Qdrant will default to using `mmap` for disk I/O operations.
For more information and benchmarks comparing io_uring with traditional I/O approaches like mmap, check out [Qdrant's io_uring implementation article.](https://qdrant.tech/articles/io_uring/) For more information and benchmarks comparing io_uring with traditional I/O approaches like mmap, check out [Qdrant's io_uring implementation article.](https://qdrant.tech/articles/io_uring/)
### Compare Results With and Without Quantization ## Performance of Quantized vs. Non-Quantized Data
Qdrant uses the quantizes vectors by default if they are avaliable. If you want to see how quantization affects your search results, you can disable it temporarily to compare results from quantized and non-quantized searches. Just set `ignore: true` in the query:
Qdrant uses the quantized vectors by default if they are available. If you want to evaluate how quantization affects your search results, you can temporarily disable it to compare results from quantized and non-quantized searches. To do this, set `ignore: true` in the query:
```http ```http
POST /collections/{collection_name}/points/query POST /collections/{collection_name}/points/query
@@ -465,14 +449,11 @@ client.query_points(
), ),
) )
``` ```
### Change the Quantization Method ### Switching Between Quantization Methods
Not sure if you’ve chosen the right quantization method? In Qdrant, you have the flexibility to remove quantization and rely solely on the original vectors, adjust the quantization type, or change compression parameters at any time without affecting your original vectors. Not sure if you’ve chosen the right quantization method? In Qdrant, you have the flexibility to remove quantization and rely solely on the original vectors, adjust the quantization type, or change compression parameters at any time without affecting your original vectors.
To switch to binary quantization and adjust the compression rate, for example, you can update the collection’s quantization configuration using the `update_collection` method:
To switch to binary quantization and adjust compression rate, for example, you can update the collection's quantization configuration using the `update_collection` method:
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -503,10 +484,8 @@ client.update_collection(
) )
``` ```
If you decide to **turn off quantization** and use only the original vectors, you can remove the quantization settings entirely with `quantization_config=None`: If you decide to **turn off quantization** and use only the original vectors, you can remove the quantization settings entirely with `quantization_config=None`:
```http ```http
PUT /collections/my_collection PUT /collections/my_collection
{ {
@@ -524,7 +503,7 @@ client.update_collection(
quantization_config=None # Remove quantization and rely on original vectors only quantization_config=None # Remove quantization and rely on original vectors only
) )
``` ```
# Wrapping Up ## Wrapping Up
![](/articles_data/what-is-vector-quantization/astronaut-running.jpg) ![](/articles_data/what-is-vector-quantization/astronaut-running.jpg)
@@ -536,18 +515,17 @@ Here are some final thoughts to help you choose the right quantization method fo
|--------------------------|-------------------------------------------------------------|--------------------------------------------------------------------------------------------| |--------------------------|-------------------------------------------------------------|--------------------------------------------------------------------------------------------|
| **Binary Quantization** | • **Fastest method and most memory-efficient**<br>• Up to **40x** faster search and **32x** reduced memory footprint | • Use with tested models like OpenAI's `text-embedding-ada-002` and Cohere's `embed-english-v2.0`<br>• When speed and memory efficiency are critical | | **Binary Quantization** | • **Fastest method and most memory-efficient**<br>• Up to **40x** faster search and **32x** reduced memory footprint | • Use with tested models like OpenAI's `text-embedding-ada-002` and Cohere's `embed-english-v2.0`<br>• When speed and memory efficiency are critical |
| **Scalar Quantization** | • **Minimal loss of accuracy**<br>• Up to **4x** reduced memory footprint | • Safe default choice for most applications.<br>• Offers a good balance between accuracy, speed, and compression. | | **Scalar Quantization** | • **Minimal loss of accuracy**<br>• Up to **4x** reduced memory footprint | • Safe default choice for most applications.<br>• Offers a good balance between accuracy, speed, and compression. |
| **Product Quantization** | • **Highest compression ratio**<br>• Up to **64x** reduced memory footprint | • When minimizing memory usage is the top priority<br>• Acceptable if some loss of accuracy is tolerable | | **Product Quantization** | • **Highest compression ratio**<br>• Up to **64x** reduced memory footprint | • When minimizing memory usage is the top priority<br>• Acceptable if some loss of accuracy and slower indexing is tolerable |
### Learn More
If you want to learn more about improving accuracy, memory efficiency, and speed when using quantization in Qdrant, we have a dedicated [Quantization tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section in our docs that explains all the quantization tips you can use to enhance your results. If you want to learn more about improving accuracy, memory efficiency, and speed when using quantization in Qdrant, we have a dedicated [Quantization tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section in our docs that explains all the quantization tips you can use to enhance your results.
Learn more about optimizing real-time precision with oversampling in Binary Quantization by watching this interview with Qdrant’s CTO, Andrey Vasnetsov: Learn more about optimizing real-time precision with oversampling in Binary Quantization by watching this interview with Qdrant’s CTO, Andrey Vasnetsov:
Continue learning about how to **optimize precision in real-time using oversampling in Binary Quantization** by watching this insightful video from Qdrant's CTO, Andrey Vasnetsov:
<div style="position: relative; padding-bottom: 56.25%; height: 0; overflow: hidden;"> <div style="position: relative; padding-bottom: 56.25%; height: 0; overflow: hidden;">
<iframe src="https://www.youtube.com/embed/4aUq5VnR_VI" frameborder="0" allowfullscreen style="position: absolute; top: 0; left: 0; width: 100%; height: 90%;"> <iframe src="https://www.youtube.com/embed/4aUq5VnR_VI" frameborder="0" allowfullscreen style="position: absolute; top: 0; left: 0; width: 100%; height: 90%;">
</iframe> </iframe>
</div> </div>
Stay up-to-date on the latest in vector search and quantization, share your projects, ask questions, [join our vector search community](https://discord.com/invite/qdrant)! Stay up-to-date on the latest in [vector search](/advanced-search/) and quantization, share your projects, ask questions, [join our vector search community](https://discord.com/invite/qdrant)!
@@ -50,7 +50,7 @@ We're incredibly excited about this collaboration with Azure Marketplace and the
Ready to elevate your business with Qdrant? **Click the banner and get started today!** Ready to elevate your business with Qdrant? **Click the banner and get started today!**
[![Get Started on Azure Marketplace](cta.png)](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db) [![Get Started on Azure Marketplace](/blog/azure-marketplace/cta.png)](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db)
### About Qdrant: ### About Qdrant:
@@ -0,0 +1,269 @@
---
title: "Qdrant 1.12 - Distance Matrix, Facet Counting & On-Disk Indexing"
draft: false
short_description: "On-Disk Text & Geo Index. Distance Matrix API. Facet API for Cardinality."
description: "Uncover insights with the Distance Matrix API, dynamically filter via Facet API, and offload additional payload to disk."
preview_image: /blog/qdrant-1.12.x/social_preview.png
social_preview_image: /blog/qdrant-1.12.x/social_preview.png
date: 2024-10-08T00:00:00-08:00
author: David Myriel
featured: true
tags:
- vector search
- distance matrix
- dimensionality reduction
- data exploration
- data visualization
- faceting
- facet api
---
[**Qdrant 1.12.0 is out!**](https://github.com/qdrant/qdrant/releases/tag/v1.12.0) Let's look at major new features and a few minor additions:
**Distance Matrix API:** Efficiently calculate pairwise distances between vectors.</br>
**GUI Data Exploration** Visually navigate your dataset and analyze vector relationships.</br>
**Faceting API:** Dynamically aggregate and count unique values in specific fields.</br>
**Text Index on disk:** Reduce memory usage by storing text indexing data on disk.</br>
**Geo Index on disk:** Offload indexed geographic data on disk for memory efficiency.
## Distance Matrix API for Data Insights
![distance-matrix-api](/blog/qdrant-1.12.x/distance-matrix-api.png)
> **Qdrant** is a similarity search engine. Our mission is to give you the tools to **discover and understand connections** between vast amounts of semantically relevant data
The **Distance Matrix API** is here to lay the groundwork for such tools.
In data exploration, tasks like [**clustering**](https://en.wikipedia.org/wiki/DBSCAN) and [**dimensionality reduction**](https://en.wikipedia.org/wiki/Dimensionality_reduction) rely on calculating distances between data points.
**Use Case:** A retail company with 10,000 customers wants to segment them by purchasing behavior. Each customer is stored as a vector in Qdrant, but without a dedicated API, clustering would need 10,000 separate batch requests, making the process inefficient and costly.
You can use this API to compute a **sparse matrix of distances** that is optimized for large datasets. Then, you can filter through the retrieved data to find the exact vector relationships that matter.
In terms of endpoints, we offer two different formats to show results:
- **Pairs** are simple, intutitive and ideal for graph representation.
- **Offsets** are more complex, but also native when defining CSR sparse matrices.
### Output - Pairs
Use the `pairs` endpoint to compare 10 random point pairs from your dataset:
```http
POST /collections/{collection_name}/points/search/matrix/pairs
{
"sample": 10,
"limit": 2
}
```
Configuring the `sample` will retrieve a random group of 10 points to compare. The `limit` is the number of semantic connections between points to consider.
Qdrant will list a sparse matrix of distances **between the closest pairs**:
```http
{
"result": {
"pairs": [
{"a": 1, "b": 3, "score": 1.4063001},
{"a": 1, "b": 4, "score": 1.2531},
{"a": 2, "b": 1, "score": 1.1550001},
{"a": 2, "b": 8, "score": 1.1359},
{"a": 3, "b": 1, "score": 1.4063001},
{"a": 3, "b": 4, "score": 1.2218001},
{"a": 4, "b": 1, "score": 1.2531},
{"a": 4, "b": 3, "score": 1.2218001},
{"a": 5, "b": 3, "score": 0.70239997},
{"a": 5, "b": 1, "score": 0.6146},
{"a": 6, "b": 3, "score": 0.6353},
{"a": 6, "b": 4, "score": 0.5093},
{"a": 7, "b": 3, "score": 1.0990001},
{"a": 7, "b": 1, "score": 1.0349001},
{"a": 8, "b": 2, "score": 1.1359},
{"a": 8, "b": 3, "score": 1.0553}
]
}
}
```
### Output - Offsets
The `offsets` endpoint offer another format of showing the distance between points:
```http
POST /collections/{collection_name}/points/search/matrix/offsets
{
"sample": 10,
"limit": 2
}
```
Qdrant will return a compact representation of the distances between points in the **form of row and column offsets**.
Two arrays, `offsets_row` and `offsets_col`, represent the positions of non-zero distance values in the matrix. Each entry in these arrays corresponds to a pair of points with a calculated distance.
```http
{
"result": {
"offsets_row": [0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6, 6, 7, 7],
"offsets_col": [2, 3, 0, 7, 0, 3, 0, 2, 2, 0, 2, 3, 2, 0, 1, 2],
"scores": [
1.4063001, 1.2531, 1.1550001, 1.1359, 1.4063001,
1.2218001, 1.2531, 1.2218001, 0.70239997, 0.6146, 0.6353,
0.5093, 1.0990001, 1.0349001, 1.1359, 1.0553
],
"ids": [1, 2, 3, 4, 5, 6, 7, 8]
}
}
```
*To learn more about the distance matrix, read [**The Distance Matrix documentation**](/documentation/concepts/explore/#distance-matrix).*
## Distance Matrix API in the Graph UI
We are adding more visualization options to the [**Graph Exploration Tool**](/blog/qdrant-1.11.x/#web-ui-graph-exploration-tool), introduced in v.1.11.
You can now leverage the **Distance Matrix API** from within this tool for a **clearer picture** of your data and its relationships.
**Example:** You can retrieve 900 `sample` points, with a `limit` of 5 connections per vector and a `tree` visualization:
```json
{
"limit": 5,
"sample": 900,
"tree": true
}
```
The new graphing method is cleaner and reveals **relationships and outliers:**
![distance-matrix](/blog/qdrant-1.12.x/distance-matrix.png)
*To learn more about the Web UI Dashboard, read the [**Interfaces documentation**](/documentation/interfaces/web-ui/).*
## Facet API for Metadata Cardinality
![facet-api](/blog/qdrant-1.12.x/facet-api.png)
In modern applications like e-commerce, users often rely on [**filters**](/articles/vector-search-filtering/), such as **brand** or **color**, to refine search results. The **Facet API** is designed to help users understand the distribution of values in a dataset.
The `facet` endpoint can efficiently count and aggregate values for a specific [**payload field**](/documentation/concepts/payload/) in your dataset.
You can use it to retrieve unique values for a field, along with the number of points that contain each value. This functionality is similar to `GROUP BY` with `COUNT(*)` in SQL databases.
> **Note:** Facet counting can only be applied to fields that support `match` conditions, such as fields with a keyword index.
### Configuration
Here’s a sample query using the REST API to facet on the `size` field, filtered by products where the `color` is red:
```http
POST /collections/{collection_name}/facet
{
"key": "size",
"filter": {
"must": {
"key": "color",
"match": { "value": "red" }
}
}
}
```
This returns counts for each unique value in the `size` field, filtered by `color` = `red`:
```json
{
"response": {
"hits": [
{"value": "L", "count": 19},
{"value": "S", "count": 10},
{"value": "M", "count": 5},
{"value": "XL", "count": 1},
{"value": "XXL", "count": 1}
]
},
"time": 0.0001
}
```
The results are sorted by count in descending order and only values with non-zero counts are returned.
### Configuration - Precise Facet
By default, facet counting runs an approximate filter. If you need a precise count, you can enable the `exact` parameter:
```http
POST /collections/{collection_name}/facet
{
"key": "size",
"exact": true
}
```
This feature provides flexibility between performance and precision, depending on the needs of your application.
*To learn more about faceting, read the [**Facet API documentation**](/documentation/concepts/payload/#facet-counts).*
## Text Index on Disk Support
![text-index-disk](/blog/qdrant-1.12.x/text-index-disk.png)
[**Qdrant text indexing**](/documentation/concepts/indexing/#full-text-index) tokenizes text into smaller units (tokens) based on chosen settings (e.g., tokenizer type, token length). These tokens are stored in an inverted index for fast text searches.
> With `on_disk` text indexing, the inverted index is stored on disk, reducing memory usage.
### Configuration
Just like with other indexes, simply add `on_disk: true` when creating the index:
```http
PUT /collections/{collection_name}/index
{
"field_name": "review_text",
"field_schema": {
"type": "text",
"tokenizer": "word",
"min_token_len": 2,
"max_token_len": 20,
"lowercase": true,
"on_disk": true
}
}
```
*To learn more about indexes, read the [**Indexing documentation**](/documentation/concepts/indexing/).*
## Geo Index on Disk Support
For [**large-scale geographic datasets**](/documentation/concepts/payload/#geo) where storing all indexes in memory is impractical, **geo indexing** allows efficient filtering of points based on geographic coordinates.
With `on_disk` geo indexing, the index is written to disk instead of residing in memory, making it possible to handle large datasets without exhausting system memory.
> This can be crucial when dealing with millions of geo points that don’t require real-time access.
### Configuration
To enable this feature, modify the index schema for the geographic field by setting the `on_disk: true` flag.
```http
PUT /collections/{collection_name}/index
{
"field_name": "location",
"field_schema": {
"type": "geo",
"on_disk": true
}
}
```
### Performance Considerations
- **Cold Query Latency:** On-disk indexes require I/O to load index segments, introducing slight latency on first access. Subsequent queries will benefit from disk caching.
- **Hot vs. Cold Indexes:** Fields frequently queried should stay in memory for faster performance, and on-disk indexes are better for large, infrequently queried fields.
- **Memory vs. Disk Trade-offs:** Users can manage memory by deciding which fields to store on disk.
![geo-index-disk](/blog/qdrant-1.12.x/geo-index-disk.png)
> To learn how to get the best performance from Qdrant, read the [**Optimization Guide**](/documentation/guides/optimize/).
## Just the Beginning
The easiest way to reach that **Hello World** moment is to [**try vector search in a live cluster**](/documentation/quickstart-cloud/). Our **interactive tutorial** will show you how to create a cluster, add data and try some filtering clauses.
**All of the new features from version 1.12 can be tested in the Web UI:**
![qdrant-filtering-tutorial](/articles_data/vector-search-filtering/qdrant-filtering-tutorial.png)
### Check Out the Tutorial Video
<iframe width="560" height="315" src="https://www.youtube.com/embed/OzTHZ0SIulg?si=yRzbgKIhwqnglawD" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
@@ -8,7 +8,7 @@ preview_image: /blog/qdrant-cpu-intel-benchmark/social_preview.jpg
social_preview_image: /blog/qdrant-cpu-intel-benchmark/social_preview.jpg social_preview_image: /blog/qdrant-cpu-intel-benchmark/social_preview.jpg
date: 2024-05-10T00:00:00-08:00 date: 2024-05-10T00:00:00-08:00
author: David Myriel, Kumar Shivendu author: David Myriel, Kumar Shivendu
featured: true featured: false
tags: tags:
- vector search - vector search
- intel benchmark - intel benchmark
@@ -0,0 +1,54 @@
---
draft: false
title: "New DeepLearning.AI Course on Retrieval Optimization: From Tokenization to Vector Quantization"
short_description: "Free, beginner-friendly course to learn retrieval optimization and boost search performance."
description: "Join Qdrant and DeepLearning.AI’s free, beginner-friendly course to learn retrieval optimization and boost search performance in machine learning."
preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg
social_preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg
date: 2024-10-06T00:02:00Z
author: Qdrant
featured: false
tags:
- DeepLearning.AI
- Vector Search
- Vector Quantization
- Tokenization
- Retrieval-Augmented Generation
- Vector Database
---
We’re excited to announce a new course on DeepLearning.AI's platform: [Retrieval Optimization: From Tokenization to Vector Quantization](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo). This collaboration between Qdrant and DeepLearning.AI aims to empower developers and data enthusiasts with the skills needed to enhance [vector search](/advanced-search/) capabilities in their applications.
Led by Qdrant’s Kacper Łukawski, this free, one-hour course is designed for beginners eager to delve into the world of retrieval optimization.
## Why This Collaboration Matters
At Qdrant, we believe in the power of effective search to transform user experiences. Partnering with DeepLearning.AI allows us to combine our cutting-edge vector search technology with their educational expertise, providing learners with a comprehensive understanding of how to build and optimize [Retrieval-Augmented Generation (RAG)](/rag/rag-evaluation-guide/) applications. This course is part of our commitment to equip the community with practical skills that leverage advanced machine learning techniques.
<iframe width="560" height="315" src="https://www.youtube.com/embed/AE8i69Kcodc?si=IdTEKlUHVbGzgJD-" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
## What You’ll Learn
In this course, you’ll explore key concepts that will enhance your understanding of retrieval optimization:
- Learn how tokenization works in large language and embedding models and how the tokenizer can affect the quality of your search.
- Explore how different tokenization techniques including Byte-Pair Encoding, WordPiece, and Unigram are trained and work.
- Understand how to [measure the quality of your retrieval](/rag/rag-evaluation-guide/) and how to optimize your search by adjusting HNSW parameters and [vector quantizations](/articles/what-is-vector-quantization/).
## Who Should Enroll
This course is tailored for anyone with basic Python knowledge.
Whether you’re starting your journey in machine learning or looking to enhance your existing skills, this course offers valuable insights to boost your capabilities.
### At a Glance:
- **Speaker**: Kacper Łukawski, Qdrant Developer Advocate
- **Level**: Beginner
- **Cost**: Free
- **Location**: Online
- **Duration**: 1 Hour
## How to Enroll
[Enroll via the DeepLearning.AI website](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo).
@@ -0,0 +1,84 @@
---
draft: false
title: "Introducing Qdrant for Startups"
short_description: "Join our Startup Program now and scale your AI-driven applications with ease."
description: "Enjoy special discounts from Qdrant, HuggingFace, LlamaIndex, and Airbyte, as well as expert support & tooling perks, and be the first to try new features."
preview_image: /blog/qdrant-for-startups-launch/preview.png
social_preview_image: /blog/qdrant-for-startups-launch/preview.png
date: 2024-10-02T00:02:00Z
author: Qdrant
featured: false
tags:
- Qdrant
- Startups
- Vector Search
- Vector Database
---
# Supporting Early-Stage Startups
Over the past few years, we’ve witnessed some of the most innovative AI applications being built on Qdrant. A significant number of these have come from startups pushing the boundaries of what’s possible in AI. To ensure these pioneering teams have access to the right resources at the right time, we're introducing **Qdrant for Startups**. This initiative is designed to provide startups with the technical support, guidance, and infrastructure they need to scale their AI innovations quickly and effectively.
Qdrant for Startups helps early-stage startups fully leverage the capabilities of vector search technology. Whether you're building retrieval-augmented generation (RAG) systems, recommendation engines, or anomaly detection models, the program offers exclusive benefits, such as discounts for Qdrant cloud, expert technical guidance, exclusive partner benefits, and co-marketing opportunities - empowering you to build and scale your AI products efficiently and cost-effectively.
## Benefits for admitted startups:
- **Qdrant Cloud discount:** 20% discount on Qdrant Cloud valid for 12 months, optimizing costs while scaling with advanced vector search capabilities.
- **Expert technical guidance:** Dedicated technical support and guidance to optimize your application’s performance with vector search.
- **Co-marketing opportunities:** Collaboration with the Qdrant team on joint marketing initiatives to boost your startup’s visibility.
- **Early access to features:** Exclusive early access to upcoming Qdrant features, keeping you at the forefront of technological advancements.
- **Community access:** Access to Qdrant’s developer and AI community for collaboration, networking, and shared learning.
## Access to popular AI tools
We’ve built this program to support startups with their entire AI tech stack. In addition to Qdrant, accepted startups will receive exclusive discounts from our program partners - Hugging Face, LlamaIndex, and Airbyte - ensuring you have access to the key tools and resources needed to build and scale AI-driven applications.
Accepted startup program members will have the ability to get additional benefits:
- Hugging Face: $100 compute credits for the HuggingFace Hub
- LlamaIndex: 20% discount for 12 months for LlamaCloud
- Airbyte: Cloud credits for Y Combinator startups
[![qdrant-for-startups-launch](/blog/qdrant-for-startups-launch/startup-cta.png)](https://qdrant.tech/qdrant-for-startups)
## Frequently Asked Questions:
**Q: What are the eligibility requirements?**
A: You must meet all of the following:
- Pre-seed, Seed or Series A startups (under five years old)
- New user of Qdrant Cloud
- Has not previously participated in the Qdrant for Startups program
- Offer is not valid to existing Qdrant customers
- Must be building an AI-driven product or services (agencies or devshops are not eligible)
- A live, functional website is required
- Billing must be done directly with Qdrant (not through a marketplace)
**Q: How can I apply to the Qdrant Startup Program?**
A: Apply through our online form by providing details about your startup and its plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities.
**Q: What criteria are used to select startups for the program?**
A: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected.
**Q: How long is the discount valid, and are there any conditions?**
A: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount.
**Q: How can I maximize the co-marketing opportunities offered by the program?**
A: Engage actively with our marketing team for features on social media, possible appearances in Discord talks or webinars, and case studies to maximize your startup's visibility and showcase your innovative use of Qdrant.
**Q: Can existing Qdrant customers apply for the Startup Program?**
A: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered by the program.
**Q: Can I reapply if my application is initially rejected?**
A: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply.
**Q: Who can I contact for more information about the program?**
A: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com.
@@ -130,7 +130,7 @@ I'm really excited to show the power of the Qdrant as vector database. Especiall
We are happy to welcome this group of people who are deeply committed to advancing vector search technology. We look forward to supporting their vision, and helping them make a bigger impact on the community. We are happy to welcome this group of people who are deeply committed to advancing vector search technology. We look forward to supporting their vision, and helping them make a bigger impact on the community.
You can find and chat with them at our [Discord Community](discord.gg/qdrant). You can find and chat with them at our [Discord Community](https://discord.gg/qdrant/).
### Why become a Qdrant Star? ### Why become a Qdrant Star?
@@ -8,7 +8,7 @@ preview_image: /blog/series-A-funding-round/series-A.png
social_preview_image: /blog/series-A-funding-round/series-A.png social_preview_image: /blog/series-A-funding-round/series-A.png
date: 2024-01-23T09:00:00.000Z date: 2024-01-23T09:00:00.000Z
author: Andre Zayarni, CEO & Co-Founder author: Andre Zayarni, CEO & Co-Founder
featured: true featured: false
tags: tags:
- Funding - Funding
- Series-A - Series-A
@@ -82,11 +82,11 @@ Here is how you can take a snapshot and recover a collection:
- For a single node cluster, call the snapshot endpoint on the exposed URL. - For a single node cluster, call the snapshot endpoint on the exposed URL.
- For a multi node cluster call a snapshot on each node of the collection. - For a multi node cluster call a snapshot on each node of the collection.
Specifically, prepend `node-{num}-` to your cluster URL. Specifically, prepend `node-{num}-` to your cluster URL.
Then call the [snapshot endpoint](../../concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0. Then call the [snapshot endpoint](/documentation/concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0.
- In the response, you'll see the name of the snapshot. - In the response, you'll see the name of the snapshot.
2. Delete and recreate the collection. 2. Delete and recreate the collection.
3. Recover the snapshot: 3. Recover the snapshot:
- Call the [recover endpoint](../../concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host. - Call the [recover endpoint](/documentation/concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host.
## Backup considerations ## Backup considerations
@@ -1,71 +0,0 @@
---
title: Configure Size & Capacity
weight: 40
aliases:
- capacity
---
# Configuring Qdrant Cloud Cluster Capacity and Size
We have been asked a lot about the optimal cluster configuration to serve a number of vectors.
The only right answer is “It depends”.
It depends on a number of factors and options you can choose for your collections.
## Basic configuration
If you need to keep all vectors in memory for maximum performance, there is a very rough formula for estimating the needed memory size looks like this:
```text
memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5
```
Extra 50% is needed for metadata (indexes, point versions, etc.) as well as for temporary segments constructed during the optimization process.
If you need to have payloads along with the vectors, it is recommended to store it on the disc, and only keep [indexed fields](../../concepts/indexing/#payload-index) in RAM.
Read more about the payload storage in the [Storage](../../concepts/storage/#payload-storage) section.
## Storage focused configuration
If your priority is to serve large amount of vectors with an average search latency, it is recommended to configure [mmap storage](../../concepts/storage/#configuring-memmap-storage).
In this case vectors will be stored on the disc in memory-mapped files, and only the most frequently used vectors will be kept in RAM.
The amount of available RAM will significantly affect the performance of the search.
As a rule of thumb, if you keep 2 times less vectors in RAM, the search latency will be 2 times lower.
The speed of disks is also important. [Let us know](/documentation/support/) if you have special requirements for a high-volume search.
## Sub-groups oriented configuration
If your use case assumes that the vectors are split into multiple collections or sub-groups based on payload values,
it is recommended to configure memory-map storage.
For example, if you serve search for multiple users, but each of them has an subset of vectors which they use independently.
In this scenario only the active subset of vectors will be kept in RAM, which allows
the fast search for the most active and recent users.
In this case you can estimate required memory size as follows:
```text
memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5
```
## Disk space
Clusters that support vector search require significant disk space. If you're
running low on disk space in your cluster, you can use the UI at
[cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster.
<aside role="status">If you use the Qdrant UI to increase the disk space in your cluster, you
cannot decrease that allocation later.</aside>
If you're running low on disk space, consider the following advantages:
- Larger Datasets: Supports larger datasets. With vector search,
larger datasets can improve the relevance and quality of search results.
- Improved Indexing: Supports the use of indexing strategies such as
HNSW (Hierarchical Navigable Small World).
- Caching: Improves speed when you cache frequently accessed data on disk.
- Backups and Redundancy: Allows more frequent backups. Perhaps the most important advantage.
@@ -27,11 +27,11 @@ Vertical scaling can be an effective way to improve the performance of a cluster
In such cases, horizontal scaling may be a more effective solution. In such cases, horizontal scaling may be a more effective solution.
Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](../../guides/distributed_deployment/#sharding) section for details. Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](/documentation/guides/distributed_deployment/#sharding) section for details.
After that, you can configure, or change the amount of Qdrant database nodes within a cluster during cluster creation, or on the cluster detail page via "Scale" button. After that, you can configure, or change the amount of Qdrant database nodes within a cluster during cluster creation, or on the cluster detail page via "Scale" button.
Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](../../guides/distributed_deployment/#cluster-scaling). Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](/documentation/guides/distributed_deployment/#cluster-scaling).
Note, that it is currently not possible to horizontally scale down the cluster in the Qdrant Cloud UI. If you require a horizontal scale down, please open a support ticket. Note, that it is currently not possible to horizontally scale down the cluster in the Qdrant Cloud UI. If you require a horizontal scale down, please open a support ticket.
@@ -20,7 +20,7 @@ A free tier cluster only includes 1 single node with the following resources:
| Disk space | 4 GB | | Disk space | 4 GB |
| Nodes | 1 | | Nodes | 1 |
This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity and sizing](/documentation/cloud/capacity-sizing/). This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity Planning](/documentation/guides/capacity-planning/).
The choice of cloud providers and regions is limited. The choice of cloud providers and regions is limited.
@@ -73,7 +73,7 @@ This page shows you how to use the Qdrant Cloud Console to create a custom Qdran
1. Choose your data center region or Hybrid Cloud environment. 1. Choose your data center region or Hybrid Cloud environment.
1. Configure RAM for each node. 1. Configure RAM for each node.
> For more information, see our [**Capacity and Sizing**](/documentation/cloud/capacity-sizing/) guidance. > For more information, see our [Capacity Planning](/documentation/guides/capacity-planning/) guidance.
1. Choose the number of vCPUs per node. If you add more 1. Choose the number of vCPUs per node. If you add more
RAM, the menu provides different options for vCPUs. RAM, the menu provides different options for vCPUs.
1. Select the number of nodes you want the cluster to be deployed on. 1. Select the number of nodes you want the cluster to be deployed on.
@@ -28,7 +28,7 @@ These settings can be changed at any time by a corresponding request.
## Setting up multitenancy ## Setting up multitenancy
**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](../../tutorials/multiple-partitions/) **How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](/documentation/tutorials/multiple-partitions/)
**When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise. **When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise.
@@ -139,12 +139,12 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
In addition to the required options, you can also specify custom values for the following collection options: In addition to the required options, you can also specify custom values for the following collection options:
* `hnsw_config` - see [indexing](../indexing/#vector-index) for details. * `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details.
* `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](../storage/#versioning) * `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](/documentation/concepts/storage/#versioning)
* `optimizers_config` - see [optimizer](../optimizer/) for details. * `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details.
* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](../../guides/distributed_deployment/#sharding) section for details. * `shard_number` - which defines how many shards the collection should have. See [distributed deployment](/documentation/guides/distributed_deployment/#sharding) section for details.
* `on_disk_payload` - defines where to store payload data. If `true` - payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload. * `on_disk_payload` - defines where to store payload data. If `true` - payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload.
* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details. * `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details.
Default parameters for the optional collection parameters are defined in [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml). Default parameters for the optional collection parameters are defined in [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml).
@@ -155,7 +155,7 @@ See [schema definitions](https://api.qdrant.tech/api-reference/collections/creat
Vectors all live in RAM for very quick access. The `on_disk` parameter can be Vectors all live in RAM for very quick access. The `on_disk` parameter can be
set in the vector configuration. If true, all vectors will live on disk. This set in the vector configuration. If true, all vectors will live on disk. This
will enable the use of will enable the use of
[memmaps](../../concepts/storage/#configuring-memmap-storage), [memmaps](/documentation/concepts/storage/#configuring-memmap-storage),
which is suitable for ingesting a large amount of data. which is suitable for ingesting a large amount of data.
### Create collection from another collection ### Create collection from another collection
@@ -466,8 +466,8 @@ For rare use cases, it is possible to create a collection without any vector sto
*Available as of v1.1.1* *Available as of v1.1.1*
For each named vector you can optionally specify For each named vector you can optionally specify
[`hnsw_config`](../indexing/#vector-index) or [`hnsw_config`](/documentation/concepts/indexing/#vector-index) or
[`quantization_config`](../../guides/quantization/#setting-up-quantization-in-qdrant) to [`quantization_config`](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) to
deviate from the collection configuration. This can be useful to fine-tune deviate from the collection configuration. This can be useful to fine-tune
search performance on a vector level. search performance on a vector level.
@@ -476,7 +476,7 @@ search performance on a vector level.
Vectors all live in RAM for very quick access. On a per-vector basis you can set Vectors all live in RAM for very quick access. On a per-vector basis you can set
`on_disk` to true to store all vectors on disk at all times. This will enable `on_disk` to true to store all vectors on disk at all times. This will enable
the use of the use of
[memmaps](../../concepts/storage/#configuring-memmap-storage), [memmaps](/documentation/concepts/storage/#configuring-memmap-storage),
which is suitable for ingesting a large amount of data. which is suitable for ingesting a large amount of data.
@@ -752,7 +752,7 @@ Outside of a unique name, there are no required configuration parameters for spa
The distance function for sparse vectors is always `Dot` and does not need to be specified. The distance function for sparse vectors is always `Dot` and does not need to be specified.
However, there are optional parameters to tune the underlying [sparse vector index](../indexing/#sparse-vector-index). However, there are optional parameters to tune the underlying [sparse vector index](/documentation/concepts/indexing/#sparse-vector-index).
### Check collection existence ### Check collection existence
@@ -928,9 +928,9 @@ client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{
The following parameters can be updated: The following parameters can be updated:
* `optimizers_config` - see [optimizer](../optimizer/) for details. * `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details.
* `hnsw_config` - see [indexing](../indexing/#vector-index) for details. * `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details.
* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details. * `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details.
* `vectors` - vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings. * `vectors` - vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings.
* `params` - other collection parameters, including `write_consistency_factor` and `on_disk_payload`. * `params` - other collection parameters, including `write_consistency_factor` and `on_disk_payload`.
@@ -1495,14 +1495,14 @@ round of automatic optimizations has completed.
To clarify: these numbers don't represent the exact amount of points or vectors To clarify: these numbers don't represent the exact amount of points or vectors
you have inserted, nor does it represent the exact number of distinguishable you have inserted, nor does it represent the exact number of distinguishable
points or vectors you can query. If you want to know exact counts, refer to the points or vectors you can query. If you want to know exact counts, refer to the
[count API](../points/#counting-points). [count API](/documentation/concepts/points/#counting-points).
_Note: these numbers may be removed in a future version of Qdrant._ _Note: these numbers may be removed in a future version of Qdrant._
### Indexing vectors in HNSW ### Indexing vectors in HNSW
In some cases, you might be surprised the value of `indexed_vectors_count` is lower than `vectors_count`. This is an intended behaviour and In some cases, you might be surprised the value of `indexed_vectors_count` is lower than `vectors_count`. This is an intended behaviour and
depends on the [optimizer configuration](../optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the depends on the [optimizer configuration](/documentation/concepts/optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the
value of `indexing_threshold`(in kB). If your collection is very small or the dimensionality of the vectors is low, there might be no HNSW segment value of `indexing_threshold`(in kB). If your collection is very small or the dimensionality of the vectors is low, there might be no HNSW segment
created and `indexed_vectors_count` might be equal to `0`. created and `indexed_vectors_count` might be equal to `0`.
@@ -7,7 +7,7 @@ aliases:
# Explore the data # Explore the data
After mastering the concepts in [search](../search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning. After mastering the concepts in [search](/documentation/concepts/search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning.
## Recommendation API ## Recommendation API
@@ -171,32 +171,32 @@ await client.QueryAsync(
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Query(context.Background(), &qdrant.QueryPoints{ client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{
Positive: []*qdrant.VectorInput{ Positive: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(100)), qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
qdrant.NewVectorInputID(qdrant.NewIDNum(231)), qdrant.NewVectorInputID(qdrant.NewIDNum(231)),
}, },
Negative: []*qdrant.VectorInput{ Negative: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(718)), qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
}), }),
Filter: &qdrant.Filter{ Filter: &qdrant.Filter{
Must: []*qdrant.Condition{ Must: []*qdrant.Condition{
qdrant.NewMatch("city", "London"), qdrant.NewMatch("city", "London"),
}, },
}, },
}) })
``` ```
@@ -368,28 +368,28 @@ await client.QueryAsync(
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Query(context.Background(), &qdrant.QueryPoints{ client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{
Positive: []*qdrant.VectorInput{ Positive: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(100)), qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
qdrant.NewVectorInputID(qdrant.NewIDNum(231)), qdrant.NewVectorInputID(qdrant.NewIDNum(231)),
}, },
Negative: []*qdrant.VectorInput{ Negative: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(718)), qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
}), }),
Using: qdrant.PtrOf("image"), Using: qdrant.PtrOf("image"),
}) })
``` ```
@@ -518,44 +518,44 @@ await client.QueryAsync(
Positive = { 100, 231 }, Positive = { 100, 231 },
Negative = { 718 } Negative = { 718 }
}, },
usingVector: "image", usingVector: "image",
limit: 10, limit: 10,
lookupFrom: new LookupLocation lookupFrom: new LookupLocation
{ {
CollectionName = "{external_collection_name}", CollectionName = "{external_collection_name}",
VectorName = "{external_vector_name}", VectorName = "{external_vector_name}",
} }
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Query(context.Background(), &qdrant.QueryPoints{ client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{
Positive: []*qdrant.VectorInput{ Positive: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(100)), qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
qdrant.NewVectorInputID(qdrant.NewIDNum(231)), qdrant.NewVectorInputID(qdrant.NewIDNum(231)),
}, },
Negative: []*qdrant.VectorInput{ Negative: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(718)), qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
}), }),
Using: qdrant.PtrOf("image"), Using: qdrant.PtrOf("image"),
LookupFrom: &qdrant.LookupLocation{ LookupFrom: &qdrant.LookupLocation{
CollectionName: "{external_collection_name}", CollectionName: "{external_collection_name}",
VectorName: qdrant.PtrOf("{external_vector_name}"), VectorName: qdrant.PtrOf("{external_vector_name}"),
}, },
}) })
``` ```
@@ -792,82 +792,82 @@ var client = new QdrantClient("localhost", 6334);
var filter = MatchKeyword("city", "london"); var filter = MatchKeyword("city", "london");
await client.QueryBatchAsync( await client.QueryBatchAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
queries: queries:
[ [
new QueryPoints() new QueryPoints()
{ {
CollectionName = "{collection_name}", CollectionName = "{collection_name}",
Query = new RecommendInput { Query = new RecommendInput {
Positive = { 100, 231 }, Positive = { 100, 231 },
Negative = { 718 }, Negative = { 718 },
}, },
Limit = 3, Limit = 3,
Filter = filter, Filter = filter,
}, },
new QueryPoints() new QueryPoints()
{ {
CollectionName = "{collection_name}", CollectionName = "{collection_name}",
Query = new RecommendInput { Query = new RecommendInput {
Positive = { 200, 67 }, Positive = { 200, 67 },
Negative = { 300 }, Negative = { 300 },
}, },
Limit = 3, Limit = 3,
Filter = filter, Filter = filter,
} }
] ]
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
filter := qdrant.Filter{ filter := qdrant.Filter{
Must: []*qdrant.Condition{ Must: []*qdrant.Condition{
qdrant.NewMatch("city", "London"), qdrant.NewMatch("city", "London"),
}, },
} }
client.QueryBatch(context.Background(), &qdrant.QueryBatchPoints{ client.QueryBatch(context.Background(), &qdrant.QueryBatchPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
QueryPoints: []*qdrant.QueryPoints{ QueryPoints: []*qdrant.QueryPoints{
{ {
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{
Positive: []*qdrant.VectorInput{ Positive: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(100)), qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
qdrant.NewVectorInputID(qdrant.NewIDNum(231)), qdrant.NewVectorInputID(qdrant.NewIDNum(231)),
}, },
Negative: []*qdrant.VectorInput{ Negative: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(718)), qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
}, },
), ),
Filter: &filter, Filter: &filter,
}, },
{ {
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{
Positive: []*qdrant.VectorInput{ Positive: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(200)), qdrant.NewVectorInputID(qdrant.NewIDNum(200)),
qdrant.NewVectorInputID(qdrant.NewIDNum(67)), qdrant.NewVectorInputID(qdrant.NewIDNum(67)),
}, },
Negative: []*qdrant.VectorInput{ Negative: []*qdrant.VectorInput{
qdrant.NewVectorInputID(qdrant.NewIDNum(300)), qdrant.NewVectorInputID(qdrant.NewIDNum(300)),
}, },
}, },
), ),
Filter: &filter, Filter: &filter,
}, },
}, },
}, },
) )
``` ```
@@ -1069,8 +1069,8 @@ using Qdrant.Client.Grpc;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.QueryAsync( await client.QueryAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
query: new DiscoverInput { query: new DiscoverInput {
Target = new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, Target = new float[] { 0.2f, 0.1f, 0.9f, 0.7f },
Context = new ContextInput { Context = new ContextInput {
Pairs = { Pairs = {
@@ -1085,39 +1085,39 @@ await client.QueryAsync(
} }
}, },
}, },
limit: 10 limit: 10
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Query(context.Background(), &qdrant.QueryPoints{ client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryDiscover(&qdrant.DiscoverInput{ Query: qdrant.NewQueryDiscover(&qdrant.DiscoverInput{
Target: qdrant.NewVectorInput(0.2, 0.1, 0.9, 0.7), Target: qdrant.NewVectorInput(0.2, 0.1, 0.9, 0.7),
Context: &qdrant.ContextInput{ Context: &qdrant.ContextInput{
Pairs: []*qdrant.ContextInputPair{ Pairs: []*qdrant.ContextInputPair{
{ {
Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(100)), Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(718)), Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
{ {
Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(200)), Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(200)),
Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(300)), Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(300)),
}, },
}, },
}, },
}), }),
}) })
``` ```
@@ -1289,30 +1289,30 @@ await client.QueryAsync(
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Query(context.Background(), &qdrant.QueryPoints{ client.Query(context.Background(), &qdrant.QueryPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Query: qdrant.NewQueryContext(&qdrant.ContextInput{ Query: qdrant.NewQueryContext(&qdrant.ContextInput{
Pairs: []*qdrant.ContextInputPair{ Pairs: []*qdrant.ContextInputPair{
{ {
Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(100)), Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(100)),
Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(718)), Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(718)),
}, },
{ {
Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(200)), Positive: qdrant.NewVectorInputID(qdrant.NewIDNum(200)),
Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(300)), Negative: qdrant.NewVectorInputID(qdrant.NewIDNum(300)),
}, },
}, },
}), }),
}) })
``` ```
@@ -1324,3 +1324,343 @@ Notes about context search:
* Best possible score is `0.0`, and it is normal that many points get this score. * Best possible score is `0.0`, and it is normal that many points get this score.
</aside> </aside>
## Distance Matrix
*Available as of v1.12.0*
The distance matrix API allows to calculate the distance between sampled pairs of vectors and to return the result as a sparse matrix.
Such API enables new data exploration use cases such as clustering similar vectors, visualization of connections or dimension reduction.
The API input request consists of the following parameters:
- `sample`: the number of vectors to sample
- `limit`: the number of scores to return per sample
- `filter`: the filter to apply to constraint the samples
Let's have a look at a basic example with `sample=100`, `limit=10`:
The engine starts by selecting `100` random points from the collection, then for each of the selected points, it will compute the top `10` closest points **within** the samples.
This will results in a total of 1000 scores represented as a sparse matrix for efficient processing.
The distance matrix API offers two output formats to ease the integration with different tools.
### Pairwise format
Returns the distance matrix as a list of pairs of point `ids` with their respective score.
```http
POST /collections/{collection_name}/points/search/matrix/pairs
{
"sample": 10,
"limit": 2,
"filter": {
"must": {
"key": "color",
"match": { "value": "red" }
}
}
}
```
```python
from qdrant_client import QdrantClient, models
client.search_matrix_pairs(
collection_name="{collection_name}",
sample=10,
limit=2,
query_filter=models.Filter(
must=[
models.FieldCondition(
key="color", match=models.MatchValue(value="red")
),
]
),
)
```
```java
import static io.qdrant.client.ConditionFactory.matchKeyword;
import io.qdrant.client.QdrantClient;
import io.qdrant.client.QdrantGrpcClient;
import io.qdrant.client.grpc.Points.Filter;
import io.qdrant.client.grpc.Points.SearchMatrixPoints;
QdrantClient client =
new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
client
.searchMatrixPairsAsync(
Points.SearchMatrixPoints.newBuilder()
.setCollectionName(collectionName)
.setFilter(Filter.newBuilder().addMust(matchKeyword("color", "red")).build())
.setSample(10)
.setLimit(2)
.build())
.get();
```
```rust
use qdrant_client::qdrant::{Condition, Filter, SearchMatrixPointsBuilder};
use qdrant_client::Qdrant;
client
.search_matrix_pairs(
SearchMatrixPointsBuilder::new("collection_name")
.filter(Filter::must(vec![Condition::matches(
"color",
"red".to_string(),
)]))
.sample(10)
.limit(2),
)
.await?;
```
```typescript
import { QdrantClient } from "@qdrant/js-client-rest";
const client = new QdrantClient({ host: "localhost", port: 6333 });
client.searchMatrixPairs("{collection_name}", {
filter: {
must: [
{
key: "color",
match: {
value: "red",
},
},
],
},
sample: 10,
limit: 2,
});
```
```csharp
using Qdrant.Client;
using Qdrant.Client.Grpc;
using static Qdrant.Client.Grpc.Conditions;
var client = new QdrantClient("localhost", 6334);
await client.SearchMatrixPairs(
collectionName: "{collection_name}",
filter: MatchKeyword("color", "red"),
sample: 10,
limit: 2
);
```
```go
import (
"context"
"github.com/qdrant/go-client/qdrant"
)
client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost",
Port: 6334,
})
sample := uint64(10)
limit := uint64(2)
res, err := client.SearchMatrixPairs(ctx, &qdrant.SearchMatrixPoints{
CollectionName: "{collection_name}",
Sample: &sample,
Limit: &limit,
Filter: &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("color", "red"),
},
},
})
```
Returns
```json
{
"result": {
"pairs": [
{"a": 1, "b": 3, "score": 1.4063001},
{"a": 1, "b": 4, "score": 1.2531},
{"a": 2, "b": 1, "score": 1.1550001},
{"a": 2, "b": 8, "score": 1.1359},
{"a": 3, "b": 1, "score": 1.4063001},
{"a": 3, "b": 4, "score": 1.2218001},
{"a": 4, "b": 1, "score": 1.2531},
{"a": 4, "b": 3, "score": 1.2218001},
{"a": 5, "b": 3, "score": 0.70239997},
{"a": 5, "b": 1, "score": 0.6146},
{"a": 6, "b": 3, "score": 0.6353},
{"a": 6, "b": 4, "score": 0.5093},
{"a": 7, "b": 3, "score": 1.0990001},
{"a": 7, "b": 1, "score": 1.0349001},
{"a": 8, "b": 2, "score": 1.1359},
{"a": 8, "b": 3, "score": 1.0553}
]
}
}
```
### Offset format
Returns the distance matrix as a four arrays:
- `offsets_row` and `offsets_col`, represent the positions of non-zero distance values in the matrix.
- `scores` contains the distance values.
- `ids` contains the point ids corresponding to the distance values.
```http
POST /collections/{collection_name}/points/search/matrix/offsets
{
"sample": 10,
"limit": 2,
"filter": {
"must": {
"key": "color",
"match": { "value": "red" }
}
}
}
```
```python
from qdrant_client import QdrantClient, models
client.search_matrix_pairs(
collection_name="{collection_name}",
sample=10,
limit=2,
query_filter=models.Filter(
must=[
models.FieldCondition(
key="color", match=models.MatchValue(value="red")
),
]
),
)
```
```java
import static io.qdrant.client.ConditionFactory.matchKeyword;
import io.qdrant.client.QdrantClient;
import io.qdrant.client.QdrantGrpcClient;
import io.qdrant.client.grpc.Points.Filter;
import io.qdrant.client.grpc.Points.SearchMatrixPoints;
QdrantClient client =
new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
client
.searchMatrixOffsetsAsync(
SearchMatrixPoints.newBuilder()
.setCollectionName(collectionName)
.setFilter(Filter.newBuilder().addMust(matchKeyword("color", "red")).build())
.setSample(10)
.setLimit(2)
.build())
.get();
```
```rust
use qdrant_client::qdrant::{Condition, Filter, SearchMatrixPointsBuilder};
use qdrant_client::Qdrant;
client
.search_matrix_offsets(
SearchMatrixPointsBuilder::new("collection_name")
.filter(Filter::must(vec![Condition::matches(
"color",
"red".to_string(),
)]))
.sample(10)
.limit(2),
)
.await?;
```
```typescript
import { QdrantClient } from "@qdrant/js-client-rest";
const client = new QdrantClient({ host: "localhost", port: 6333 });
client.searchMatrixOffsets("{collection_name}", {
filter: {
must: [
{
key: "color",
match: {
value: "red",
},
},
],
},
sample: 10,
limit: 2,
});
```
```csharp
using Qdrant.Client;
using Qdrant.Client.Grpc;
using static Qdrant.Client.Grpc.Conditions;
var client = new QdrantClient("localhost", 6334);
await client.SearchMatrixOffsets(
collectionName: "{collection_name}",
filter: MatchKeyword("color", "red"),
sample: 10,
limit: 2
);
```
```go
import (
"context"
"github.com/qdrant/go-client/qdrant"
)
client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost",
Port: 6334,
})
sample := uint64(10)
limit := uint64(2)
res, err := client.SearchMatrixOffsets(ctx, &qdrant.SearchMatrixPoints{
CollectionName: "{collection_name}",
Sample: &sample,
Limit: &limit,
Filter: &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("color", "red"),
},
},
})
```
Returns
```json
{
"result": {
"offsets_row": [0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6, 6, 7, 7],
"offsets_col": [2, 3, 0, 7, 0, 3, 0, 2, 2, 0, 2, 3, 2, 0, 1, 2],
"scores": [
1.4063001, 1.2531, 1.1550001, 1.1359, 1.4063001,
1.2218001, 1.2531, 1.2218001, 0.70239997, 0.6146, 0.6353,
0.5093, 1.0990001, 1.0349001, 1.1359, 1.0553
],
"ids": [1, 2, 3, 4, 5, 6, 7, 8]
}
}
```
@@ -8,7 +8,7 @@ aliases:
# Filtering # Filtering
With Qdrant, you can set conditions when searching or retrieving points. With Qdrant, you can set conditions when searching or retrieving points.
For example, you can impose conditions on both the [payload](../payload/) and the `id` of the point. For example, you can impose conditions on both the [payload](/documentation/concepts/payload/) and the `id` of the point.
Setting additional conditions is important when it is impossible to express all the features of the object in the embedding. Setting additional conditions is important when it is impossible to express all the features of the object in the embedding.
Examples include a variety of business requirements: stock availability, user location, or desired price range. Examples include a variety of business requirements: stock availability, user location, or desired price range.
@@ -838,7 +838,7 @@ qdrant.NewMatchInt("count", 0)
The simplest kind of condition is one that checks if the stored value equals the given one. The simplest kind of condition is one that checks if the stored value equals the given one.
If several values are stored, at least one of them should match the condition. If several values are stored, at least one of them should match the condition.
You can apply it to [keyword](../payload/#keyword), [integer](../payload/#integer) and [bool](../payload/#bool) payloads. You can apply it to [keyword](/documentation/concepts/payload/#keyword), [integer](/documentation/concepts/payload/#integer) and [bool](/documentation/concepts/payload/#bool) payloads.
### Match Any ### Match Any
@@ -847,7 +847,7 @@ You can apply it to [keyword](../payload/#keyword), [integer](../payload/#intege
In case you want to check if the stored value is one of multiple values, you can use the Match Any condition. In case you want to check if the stored value is one of multiple values, you can use the Match Any condition.
Match Any works as a logical OR for the given values. It can also be described as a `IN` operator. Match Any works as a logical OR for the given values. It can also be described as a `IN` operator.
You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads. You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads.
Example: Example:
@@ -909,7 +909,7 @@ In case you want to check if the stored value is not one of multiple values, you
Match Except works as a logical NOR for the given values. Match Except works as a logical NOR for the given values.
It can also be described as a `NOT IN` operator. It can also be described as a `NOT IN` operator.
You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads. You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads.
Example: Example:
@@ -1908,7 +1908,7 @@ A special case of the `match` condition is the `text` match condition.
It allows you to search for a specific substring, token or phrase within the text field. It allows you to search for a specific substring, token or phrase within the text field.
Exact texts that will match the condition depend on full-text index configuration. Exact texts that will match the condition depend on full-text index configuration.
Configuration is defined during the index creation and describe at [full-text index](../indexing/#full-text-index). Configuration is defined during the index creation and describe at [full-text index](/documentation/concepts/indexing/#full-text-index).
If there is no full-text index for the field, the condition will work as exact substring match. If there is no full-text index for the field, the condition will work as exact substring match.
@@ -2047,11 +2047,11 @@ Comparisons that can be used:
- `lt` - less than - `lt` - less than
- `lte` - less than or equal - `lte` - less than or equal
Can be applied to [float](../payload/#float) and [integer](../payload/#integer) payloads. Can be applied to [float](/documentation/concepts/payload/#float) and [integer](/documentation/concepts/payload/#integer) payloads.
### Datetime Range ### Datetime Range
The datetime range is a unique range condition, used for [datetime](../payload/#datetime) payloads, which supports RFC 3339 formats. The datetime range is a unique range condition, used for [datetime](/documentation/concepts/payload/#datetime) payloads, which supports RFC 3339 formats.
You do not need to convert dates to UNIX timestaps. During comparison, timestamps are parsed and converted to UTC. You do not need to convert dates to UNIX timestaps. During comparison, timestamps are parsed and converted to UTC.
_Available as of v1.8.0_ _Available as of v1.8.0_
@@ -2364,7 +2364,7 @@ qdrant.NewGeoRadius("location", 52.520711, 13.403683, 1000.0)
It matches with `location`s inside a circle with the `center` at the center and a radius of `radius` meters. It matches with `location`s inside a circle with the `center` at the center and a radius of `radius` meters.
If several values are stored, at least one of them should match the condition. If several values are stored, at least one of them should match the condition.
These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo). These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo).
#### Geo Polygon #### Geo Polygon
Geo Polygons search is useful for when you want to find points inside an irregularly shaped area, for example a country boundary or a forest boundary. A polygon always has an exterior ring and may optionally include interior rings. A lake with an island would be an example of an interior ring. If you wanted to find points in the water but not on the island, you would make an interior ring for the island. Geo Polygons search is useful for when you want to find points inside an irregularly shaped area, for example a country boundary or a forest boundary. A polygon always has an exterior ring and may optionally include interior rings. A lake with an island would be an example of an interior ring. If you wanted to find points in the water but not on the island, you would make an interior ring for the island.
@@ -2659,7 +2659,7 @@ qdrant.NewGeoPolygon("location",
A match is considered any point location inside or on the boundaries of the given polygon's exterior but not inside any interiors. A match is considered any point location inside or on the boundaries of the given polygon's exterior but not inside any interiors.
If several location values are stored for a point, then any of them matching will include that point as a candidate in the resultset. If several location values are stored for a point, then any of them matching will include that point as a candidate in the resultset.
These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo). These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo).
### Values count ### Values count
@@ -10,7 +10,7 @@ hideInSidebar: false # Optional. If true, the page will not be shown in the side
*Available as of v1.10.0* *Available as of v1.10.0*
With the introduction of [many named vectors per point](../vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries, With the introduction of [many named vectors per point](/documentation/concepts/vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries,
or by performing the search in more than one stage. or by performing the search in more than one stage.
Qdrant has a flexible and universal interface to make this possible, called `Query API` ([API reference](https://api.qdrant.tech/api-reference/search/query-points)). Qdrant has a flexible and universal interface to make this possible, called `Query API` ([API reference](https://api.qdrant.tech/api-reference/search/query-points)).
@@ -793,7 +793,7 @@ Other than the introduction of `prefetch`, the `Query API` has been designed to
### Query by ID ### Query by ID
Whenever you need to use a vector as an input, you can always use a [point ID](../points/#point-ids) instead. Whenever you need to use a vector as an input, you can always use a [point ID](/documentation/concepts/points/#point-ids) instead.
```http ```http
POST /collections/{collection_name}/points/query POST /collections/{collection_name}/points/query
@@ -1397,4 +1397,4 @@ client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{
}) })
``` ```
For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](./search/#search-groups) and [lookup](./search/#lookup-in-groups). For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](/documentation/concepts/search/#search-groups) and [lookup](/documentation/concepts/search/#lookup-in-groups).
@@ -12,14 +12,14 @@ A key feature of Qdrant is the effective combination of vector and traditional i
The indexes in the segments exist independently, but the parameters of the indexes themselves are configured for the whole collection. The indexes in the segments exist independently, but the parameters of the indexes themselves are configured for the whole collection.
Not all segments automatically have indexes. Not all segments automatically have indexes.
Their necessity is determined by the [optimizer](../optimizer/) settings and depends, as a rule, on the number of stored points. Their necessity is determined by the [optimizer](/documentation/concepts/optimizer/) settings and depends, as a rule, on the number of stored points.
## Payload Index ## Payload Index
Payload index in Qdrant is similar to the index in conventional document-oriented databases. Payload index in Qdrant is similar to the index in conventional document-oriented databases.
This index is built for a specific field and type, and is used for quick point requests by the corresponding filtering condition. This index is built for a specific field and type, and is used for quick point requests by the corresponding filtering condition.
The index is also used to accurately estimate the filter cardinality, which helps the [query planning](../search/#query-planning) choose a search strategy. The index is also used to accurately estimate the filter cardinality, which helps the [query planning](/documentation/concepts/search/#query-planning) choose a search strategy.
Creating an index requires additional computational resources and memory, so choosing fields to be indexed is essential. Qdrant does not make this choice but grants it to the user. Creating an index requires additional computational resources and memory, so choosing fields to be indexed is essential. Qdrant does not make this choice but grants it to the user.
@@ -119,19 +119,19 @@ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection
}) })
``` ```
You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](../filtering/#nested-key). You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](/documentation/concepts/filtering/#nested-key).
Available field types are: Available field types are:
* `keyword` - for [keyword](../payload/#keyword) payload, affects [Match](../filtering/#match) filtering conditions. * `keyword` - for [keyword](/documentation/concepts/payload/#keyword) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions.
* `integer` - for [integer](../payload/#integer) payload, affects [Match](../filtering/#match) and [Range](../filtering/#range) filtering conditions. * `integer` - for [integer](/documentation/concepts/payload/#integer) payload, affects [Match](/documentation/concepts/filtering/#match) and [Range](/documentation/concepts/filtering/#range) filtering conditions.
* `float` - for [float](../payload/#float) payload, affects [Range](../filtering/#range) filtering conditions. * `float` - for [float](/documentation/concepts/payload/#float) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions.
* `bool` - for [bool](../payload/#bool) payload, affects [Match](../filtering/#match) filtering conditions (available as of v1.4.0). * `bool` - for [bool](/documentation/concepts/payload/#bool) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions (available as of v1.4.0).
* `geo` - for [geo](../payload/#geo) payload, affects [Geo Bounding Box](../filtering/#geo-bounding-box) and [Geo Radius](../filtering/#geo-radius) filtering conditions. * `geo` - for [geo](/documentation/concepts/payload/#geo) payload, affects [Geo Bounding Box](/documentation/concepts/filtering/#geo-bounding-box) and [Geo Radius](/documentation/concepts/filtering/#geo-radius) filtering conditions.
* `datetime` - for [datetime](../payload/#datetime) payload, affects [Range](../filtering/#range) filtering conditions (available as of v1.8.0). * `datetime` - for [datetime](/documentation/concepts/payload/#datetime) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions (available as of v1.8.0).
* `text` - a special kind of index, available for [keyword](../payload/#keyword) / string payloads, affects [Full Text search](../filtering/#full-text-match) filtering conditions. * `text` - a special kind of index, available for [keyword](/documentation/concepts/payload/#keyword) / string payloads, affects [Full Text search](/documentation/concepts/filtering/#full-text-match) filtering conditions.
* `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](../payload/#uuid). * `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](/documentation/concepts/payload/#uuid).
Affects [Match](../filtering/#match) filtering conditions. (available as of v1.11.0) Affects [Match](/documentation/concepts/filtering/#match) filtering conditions. (available as of v1.11.0)
Payload index may occupy some additional memory, so it is recommended to only use index for those fields that are used in filtering conditions. Payload index may occupy some additional memory, so it is recommended to only use index for those fields that are used in filtering conditions.
If you need to filter by many fields and the memory limits does not allow to index all of them, it is recommended to choose the field that limits the search result the most. If you need to filter by many fields and the memory limits does not allow to index all of them, it is recommended to choose the field that limits the search result the most.
@@ -313,7 +313,7 @@ Available tokenizers are:
* `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`. * `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`.
* `multilingual` - special type of tokenizer based on [charabia](https://github.com/meilisearch/charabia) package. It allows proper tokenization and lemmatization for multiple languages, including those with non-latin alphabets and non-space delimiters. See [charabia documentation](https://github.com/meilisearch/charabia) for full list of supported languages supported normalization options. In the default build configuration, qdrant does not include support for all languages, due to the increasing size of the resulting binary. Chinese, Japanese and Korean languages are not enabled by default, but can be enabled by building qdrant from source with `--features multiling-chinese,multiling-japanese,multiling-korean` flags. * `multilingual` - special type of tokenizer based on [charabia](https://github.com/meilisearch/charabia) package. It allows proper tokenization and lemmatization for multiple languages, including those with non-latin alphabets and non-space delimiters. See [charabia documentation](https://github.com/meilisearch/charabia) for full list of supported languages supported normalization options. In the default build configuration, qdrant does not include support for all languages, due to the increasing size of the resulting binary. Chinese, Japanese and Korean languages are not enabled by default, but can be enabled by building qdrant from source with `--features multiling-chinese,multiling-japanese,multiling-korean` flags.
See [Full Text match](../filtering/#full-text-match) for examples of querying with full-text index. See [Full Text match](/documentation/concepts/filtering/#full-text-match) for examples of querying with full-text index.
### Parameterized index ### Parameterized index
@@ -636,6 +636,8 @@ Payload index on-disk is supported for following types:
* `float` * `float`
* `datetime` * `datetime`
* `uuid` * `uuid`
* `text`
* `geo`
The list will be extended in future versions. The list will be extended in future versions.
@@ -645,7 +647,7 @@ The list will be extended in future versions.
Many vector search use-cases require multitenancy. In a multi-tenant scenario the collection is expected to contain multiple subsets of data, where each subset belongs to a different tenant. Many vector search use-cases require multitenancy. In a multi-tenant scenario the collection is expected to contain multiple subsets of data, where each subset belongs to a different tenant.
Qdrant supports efficient multi-tenant search by enabling [special configuration](../guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant. Qdrant supports efficient multi-tenant search by enabling [special configuration](/documentation/guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant.
<aside role="note"> <aside role="note">
In Qdrant, tenants are not necessarily non-overlapping. It is possible to have subsets of data that belong to multiple tenants. In Qdrant, tenants are not necessarily non-overlapping. It is possible to have subsets of data that belong to multiple tenants.
@@ -960,7 +962,7 @@ storage:
``` ```
And so in the process of creating a [collection](../collections/). The `ef` parameter is configured during [the search](../search/) and by default is equal to `ef_construct`. And so in the process of creating a [collection](/documentation/concepts/collections/). The `ef` parameter is configured during [the search](/documentation/concepts/search/) and by default is equal to `ef_construct`.
HNSW is chosen for several reasons. HNSW is chosen for several reasons.
First, HNSW is well-compatible with the modification that allows Qdrant to use filters during a search. First, HNSW is well-compatible with the modification that allows Qdrant to use filters during a search.
@@ -969,7 +971,7 @@ Second, it is one of the most accurate and fastest algorithms, according to [pub
*Available as of v1.1.1* *Available as of v1.1.1*
The HNSW parameters can also be configured on a collection and named vector The HNSW parameters can also be configured on a collection and named vector
level by setting [`hnsw_config`](../indexing/#vector-index) to fine-tune search level by setting [`hnsw_config`](/documentation/concepts/indexing/#vector-index) to fine-tune search
performance. performance.
## Sparse Vector Index ## Sparse Vector Index
@@ -9,7 +9,7 @@ aliases:
It is much more efficient to apply changes in batches than perform each change individually, as many other databases do. Qdrant here is no exception. Since Qdrant operates with data structures that are not always easy to change, it is sometimes necessary to rebuild those structures completely. It is much more efficient to apply changes in batches than perform each change individually, as many other databases do. Qdrant here is no exception. Since Qdrant operates with data structures that are not always easy to change, it is sometimes necessary to rebuild those structures completely.
Storage optimization in Qdrant occurs at the segment level (see [storage](../storage/)). Storage optimization in Qdrant occurs at the segment level (see [storage](/documentation/concepts/storage/)).
In this case, the segment to be optimized remains readable for the time of the rebuild. In this case, the segment to be optimized remains readable for the time of the rebuild.
![Segment optimization](/docs/optimization.svg) ![Segment optimization](/docs/optimization.svg)
@@ -91,6 +91,6 @@ storage:
indexing_threshold_kb: 20000 indexing_threshold_kb: 20000
``` ```
In addition to the configuration file, you can also set optimizer parameters separately for each [collection](../collections/). In addition to the configuration file, you can also set optimizer parameters separately for each [collection](/documentation/concepts/collections/).
Dynamic parameter updates may be useful, for example, for more efficient initial loading of points. You can disable indexing during the upload process with these settings and enable it immediately after it is finished. As a result, you will not waste extra computation resources on rebuilding the index. Dynamic parameter updates may be useful, for example, for more efficient initial loading of points. You can disable indexing during the upload process with these settings and enable it immediately after it is finished. As a result, you will not waste extra computation resources on rebuilding the index.
@@ -46,11 +46,11 @@ This feature is implemented as additional filters during the search and will ena
During the filtering, Qdrant will check the conditions over those values that match the type of the filtering condition. If the stored value type does not fit the filtering condition - it will be considered not satisfied. During the filtering, Qdrant will check the conditions over those values that match the type of the filtering condition. If the stored value type does not fit the filtering condition - it will be considered not satisfied.
For example, you will get an empty output if you apply the [range condition](../filtering/#range) on the string data. For example, you will get an empty output if you apply the [range condition](/documentation/concepts/filtering/#range) on the string data.
However, arrays (multiple values of the same type) are treated a little bit different. When we apply a filter to an array, it will succeed if at least one of the values inside the array meets the condition. However, arrays (multiple values of the same type) are treated a little bit different. When we apply a filter to an array, it will succeed if at least one of the values inside the array meets the condition.
The filtering process is discussed in detail in the section [Filtering](../filtering/). The filtering process is discussed in detail in the section [Filtering](/documentation/concepts/filtering/).
Let's look at the data types that Qdrant supports for searching: Let's look at the data types that Qdrant supports for searching:
@@ -374,73 +374,73 @@ using Qdrant.Client.Grpc;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.UpsertAsync( await client.UpsertAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
points: new List<PointStruct> points: new List<PointStruct>
{ {
new PointStruct new PointStruct
{ {
Id = 1, Id = 1,
Vectors = new[] { 0.05f, 0.61f, 0.76f, 0.74f }, Vectors = new[] { 0.05f, 0.61f, 0.76f, 0.74f },
Payload = { ["city"] = "Berlin", ["price"] = 1.99 } Payload = { ["city"] = "Berlin", ["price"] = 1.99 }
}, },
new PointStruct new PointStruct
{ {
Id = 2, Id = 2,
Vectors = new[] { 0.19f, 0.81f, 0.75f, 0.11f }, Vectors = new[] { 0.19f, 0.81f, 0.75f, 0.11f },
Payload = { ["city"] = new[] { "Berlin", "London" } } Payload = { ["city"] = new[] { "Berlin", "London" } }
}, },
new PointStruct new PointStruct
{ {
Id = 3, Id = 3,
Vectors = new[] { 0.36f, 0.55f, 0.47f, 0.94f }, Vectors = new[] { 0.36f, 0.55f, 0.47f, 0.94f },
Payload = Payload =
{ {
["city"] = new[] { "Berlin", "Moscow" }, ["city"] = new[] { "Berlin", "Moscow" },
["price"] = new Value ["price"] = new Value
{ {
ListValue = new ListValue { Values = { new Value[] { 1.99, 2.99 } } } ListValue = new ListValue { Values = { new Value[] { 1.99, 2.99 } } }
} }
} }
} }
} }
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.Upsert(context.Background(), &qdrant.UpsertPoints{ client.Upsert(context.Background(), &qdrant.UpsertPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Points: []*qdrant.PointStruct{ Points: []*qdrant.PointStruct{
{ {
Id: qdrant.NewIDNum(1), Id: qdrant.NewIDNum(1),
Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74), Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74),
Payload: qdrant.NewValueMap(map[string]any{ Payload: qdrant.NewValueMap(map[string]any{
"city": "Berlin", "price": 1.99}), "city": "Berlin", "price": 1.99}),
}, },
{ {
Id: qdrant.NewIDNum(2), Id: qdrant.NewIDNum(2),
Vectors: qdrant.NewVectors(0.19, 0.81, 0.75, 0.11), Vectors: qdrant.NewVectors(0.19, 0.81, 0.75, 0.11),
Payload: qdrant.NewValueMap(map[string]any{ Payload: qdrant.NewValueMap(map[string]any{
"city": []any{"Berlin", "London"}}), "city": []any{"Berlin", "London"}}),
}, },
{ {
Id: qdrant.NewIDNum(3), Id: qdrant.NewIDNum(3),
Vectors: qdrant.NewVectors(0.36, 0.55, 0.47, 0.94), Vectors: qdrant.NewVectors(0.36, 0.55, 0.47, 0.94),
Payload: qdrant.NewValueMap(map[string]any{ Payload: qdrant.NewValueMap(map[string]any{
"city": []any{"Berlin", "London"}, "city": []any{"Berlin", "London"},
"price": []any{1.99, 2.99}}), "price": []any{1.99, 2.99}}),
}, },
}, },
}) })
``` ```
@@ -536,31 +536,31 @@ using Qdrant.Client.Grpc;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.SetPayloadAsync( await client.SetPayloadAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } }, payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } },
ids: new ulong[] { 0, 3, 10 } ids: new ulong[] { 0, 3, 10 }
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{ client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Payload: qdrant.NewValueMap( Payload: qdrant.NewValueMap(
map[string]any{"property1": "string", "property2": "string"}), map[string]any{"property1": "string", "property2": "string"}),
PointsSelector: qdrant.NewPointsSelector( PointsSelector: qdrant.NewPointsSelector(
qdrant.NewIDNum(0), qdrant.NewIDNum(0),
qdrant.NewIDNum(3)), qdrant.NewIDNum(3)),
}) })
``` ```
@@ -673,33 +673,33 @@ using static Qdrant.Client.Grpc.Conditions;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.SetPayloadAsync( await client.SetPayloadAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } }, payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } },
filter: MatchKeyword("color", "red") filter: MatchKeyword("color", "red")
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{ client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Payload: qdrant.NewValueMap( Payload: qdrant.NewValueMap(
map[string]any{"property1": "string", "property2": "string"}), map[string]any{"property1": "string", "property2": "string"}),
PointsSelector: qdrant.NewPointsSelectorFilter(&qdrant.Filter{ PointsSelector: qdrant.NewPointsSelectorFilter(&qdrant.Filter{
Must: []*qdrant.Condition{ Must: []*qdrant.Condition{
qdrant.NewMatch("color", "red"), qdrant.NewMatch("color", "red"),
}, },
}), }),
}) })
``` ```
@@ -833,31 +833,31 @@ using Qdrant.Client.Grpc;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.OverwritePayloadAsync( await client.OverwritePayloadAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } }, payload: new Dictionary<string, Value> { { "property1", "string" }, { "property2", "string" } },
ids: new ulong[] { 0, 3, 10 } ids: new ulong[] { 0, 3, 10 }
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.OverwritePayload(context.Background(), &qdrant.SetPayloadPoints{ client.OverwritePayload(context.Background(), &qdrant.SetPayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Payload: qdrant.NewValueMap( Payload: qdrant.NewValueMap(
map[string]any{"property1": "string", "property2": "string"}), map[string]any{"property1": "string", "property2": "string"}),
PointsSelector: qdrant.NewPointsSelector( PointsSelector: qdrant.NewPointsSelector(
qdrant.NewIDNum(0), qdrant.NewIDNum(0),
qdrant.NewIDNum(3)), qdrant.NewIDNum(3)),
}) })
``` ```
@@ -924,21 +924,21 @@ await client.ClearPayloadAsync(collectionName: "{collection_name}", ids: new ulo
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.ClearPayload(context.Background(), &qdrant.ClearPayloadPoints{ client.ClearPayload(context.Background(), &qdrant.ClearPayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Points: qdrant.NewPointsSelector( Points: qdrant.NewPointsSelector(
qdrant.NewIDNum(0), qdrant.NewIDNum(0),
qdrant.NewIDNum(3)), qdrant.NewIDNum(3)),
}) })
``` ```
@@ -1014,30 +1014,30 @@ using Qdrant.Client;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.DeletePayloadAsync( await client.DeletePayloadAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
keys: ["color", "price"], keys: ["color", "price"],
ids: new ulong[] { 0, 3, 100 } ids: new ulong[] { 0, 3, 100 }
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{ client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Keys: []string{"color", "price"}, Keys: []string{"color", "price"},
PointsSelector: qdrant.NewPointsSelector( PointsSelector: qdrant.NewPointsSelector(
qdrant.NewIDNum(0), qdrant.NewIDNum(0),
qdrant.NewIDNum(3)), qdrant.NewIDNum(3)),
}) })
``` ```
@@ -1132,32 +1132,32 @@ using static Qdrant.Client.Grpc.Conditions;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.DeletePayloadAsync( await client.DeletePayloadAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
keys: ["color", "price"], keys: ["color", "price"],
filter: MatchKeyword("color", "red") filter: MatchKeyword("color", "red")
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{ client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
Keys: []string{"color", "price"}, Keys: []string{"color", "price"},
PointsSelector: qdrant.NewPointsSelectorFilter( PointsSelector: qdrant.NewPointsSelectorFilter(
&qdrant.Filter{ &qdrant.Filter{
Must: []*qdrant.Condition{qdrant.NewMatch("color", "red")}, Must: []*qdrant.Condition{qdrant.NewMatch("color", "red")},
}, },
), ),
}) })
``` ```
@@ -1165,7 +1165,7 @@ client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{
To search more efficiently with filters, Qdrant allows you to create indexes for payload fields by specifying the name and type of field it is intended to be. To search more efficiently with filters, Qdrant allows you to create indexes for payload fields by specifying the name and type of field it is intended to be.
The indexed fields also affect the vector index. See [Indexing](../indexing/) for details. The indexed fields also affect the vector index. See [Indexing](/documentation/concepts/indexing/) for details.
In practice, we recommend creating an index on those fields that could potentially constrain the results the most. In practice, we recommend creating an index on those fields that could potentially constrain the results the most.
For example, using an index for the object ID will be much more efficient, being unique for each record, than an index by its color, which has only a few possible values. For example, using an index for the object ID will be much more efficient, being unique for each record, than an index by its color, which has only a few possible values.
@@ -1233,27 +1233,27 @@ using Qdrant.Client;
var client = new QdrantClient("localhost", 6334); var client = new QdrantClient("localhost", 6334);
await client.CreatePayloadIndexAsync( await client.CreatePayloadIndexAsync(
collectionName: "{collection_name}", collectionName: "{collection_name}",
fieldName: "name_of_the_field_to_index" fieldName: "name_of_the_field_to_index"
); );
``` ```
```go ```go
import ( import (
"context" "context"
"github.com/qdrant/go-client/qdrant" "github.com/qdrant/go-client/qdrant"
) )
client, err := qdrant.NewClient(&qdrant.Config{ client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost", Host: "localhost",
Port: 6334, Port: 6334,
}) })
client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
CollectionName: "{collection_name}", CollectionName: "{collection_name}",
FieldName: "name_of_the_field_to_index", FieldName: "name_of_the_field_to_index",
FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
}) })
``` ```
@@ -1273,3 +1273,228 @@ Payload schema example:
} }
} }
``` ```
## Facet counts
*Available as of v1.12.0*
Faceting is a special counting technique that can be used for various purposes:
- Know which unique values exist for a payload key.
- Know the number of points that contain each unique value.
- Know how restrictive a filter would become by matching a specific value.
Specifically, it is a counting aggregation for the values in a field, akin to a `GROUP BY` with `COUNT(*)` commands in SQL.
These results for a specific field is called a "facet". For example, when you look at an e-commerce search results page, you might see a list of brands on the sidebar, showing the number of products for each brand. This would be a facet for a `"brand"` field.
<aside role="status">In Qdrant you can facet on a field <strong>only</strong> if you have created a field index that supports <code>MatchValue</code> conditions for it, like a <code>keyword</code> index.</aside>
To get the facet counts for a field, you can use the following:
REST API ([Facet](https://api.qdrant.tech/api-reference/search/facet))
```http
POST /collections/{collection_name}/facet
{
"key": "size",
"filter": {
"must": {
"key": "color",
"match": { "value": "red" }
}
}
}
```
```python
from qdrant_client import QdrantClient, models
client = QdrantClient(url="http://localhost:6333")
client.facet(
collection_name="{collection_name}",
key="size",
facet_filter=models.Filter(must=[models.Match("color", "red")]),
)
```
```typescript
import { QdrantClient } from "@qdrant/js-client-rest";
const client = new QdrantClient({ host: "localhost", port: 6333 });
client.facet("{collection_name}", {
filter: {
must: [
{
key: "color",
match: {
value: "red",
},
},
],
},
key: "size",
});
```
```rust
use qdrant_client::qdrant::{Condition, FacetCountsBuilder, Filter};
use qdrant_client::Qdrant;
let client = Qdrant::from_url("http://localhost:6334").build()?;
client
.facet(
FacetCountsBuilder::new("{collection_name}", "size")
.limit(10)
.filter(Filter::must(vec![Condition::matches(
"color",
"red".to_string(),
)])),
)
.await?;
```
```java
import io.qdrant.client.QdrantClient;
import io.qdrant.client.QdrantGrpcClient;
import static io.qdrant.client.ConditionFactory.matchKeyword;
import io.qdrant.client.grpc.Points;
import io.qdrant.client.grpc.Filter;
QdrantClient client = new QdrantClient(
QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
client
.facetAsync(
Points.FacetCounts.newBuilder()
.setCollectionName(collection_name)
.setKey("size")
.setFilter(Filter.newBuilder().addMust(matchKeyword("color", "red")).build())
.build())
.get();
```
```csharp
using Qdrant.Client;
using static Qdrant.Client.Grpc.Conditions;
var client = new QdrantClient("localhost", 6334);
await client.FacetAsync(
"{collection_name}",
key: "size",
filter: MatchKeyword("color", "red"),
);
```
```go
import (
"context"
"github.com/qdrant/go-client/qdrant"
)
client, err := qdrant.NewClient(&qdrant.Config{
Host: "localhost",
Port: 6334,
})
res, err := client.Facet(ctx, &qdrant.FacetCounts{
CollectionName: "{collection_name}",
Key: "size",
Filter: &qdrant.Filter{
Must: []*qdrant.Condition{
qdrant.NewMatch("color", "red"),
},
},
})
```
The response will contain the counts for each unique value in the field:
```json
{
"response": {
"hits": [
{"value": "L", "count": 19},
{"value": "S", "count": 10},
{"value": "M", "count": 5},
{"value": "XL", "count": 1},
{"value": "XXL", "count": 1}
]
},
"time": 0.0001
}
```
The results are sorted by the count in descending order, then by the value in ascending order.
Only values with non-zero counts will be returned.
By default, the way Qdrant the counts for each value is approximate to achieve fast results. This should accurate enough for most cases, but if you need to debug your storage, you can use the `exact` parameter to get exact counts.
```http
POST /collections/{collection_name}/facet
{
"key": "size",
"exact": true
}
```
```python
client.facet(
collection_name="{collection_name}",
key="size",
exact=True,
)
```
```typescript
client.facet("{collection_name}", {
key: "size",
exact: true,
});
```
```rust
use qdrant_client::qdrant::FacetCountsBuilder;
client
.facet(
FacetCountsBuilder::new("{collection_name}", "size")
.limit(10)
.exact(true),
)
.await?;
```
```java
client
.facetAsync(
Points.FacetCounts.newBuilder()
.setCollectionName(collection_name)
.setKey("foo")
.setExact(true)
.build())
.get();
```
```csharp
using Qdrant.Client;
await client.FacetAsync(
"{collection_name}",
key: "size",
exact: true,
);
```
```go
res, err := client.Facet(ctx, &qdrant.FacetCounts{
CollectionName: "{collection_name}",
Key: "key",
Exact: true,
})
```
@@ -8,7 +8,7 @@ aliases:
# Points # Points
The points are the central entity that Qdrant operates with. The points are the central entity that Qdrant operates with.
A point is a record consisting of a [vector](../vectors/) and an optional [payload](../payload/). A point is a record consisting of a [vector](/documentation/concepts/vectors/) and an optional [payload](/documentation/concepts/payload/).
It looks like this: It looks like this:
@@ -21,8 +21,8 @@ It looks like this:
} }
``` ```
You can search among the points grouped in one [collection](../collections/) based on vector similarity. You can search among the points grouped in one [collection](/documentation/concepts/collections/) based on vector similarity.
This procedure is described in more detail in the [search](../search/) and [filtering](../filtering/) sections. This procedure is described in more detail in the [search](/documentation/concepts/search/) and [filtering](/documentation/concepts/filtering/) sections.
This section explains how to create and manage vectors. This section explains how to create and manage vectors.
@@ -343,7 +343,7 @@ Here is a list of supported vector types:
It is possible to attach more than one type of vector to a single point. It is possible to attach more than one type of vector to a single point.
In Qdrant we call it Named Vectors. In Qdrant we call it Named Vectors.
Read more about vector types, how they are stored and optimized in the [vectors](../vectors/) section. Read more about vector types, how they are stored and optimized in the [vectors](/documentation/concepts/vectors/) section.
## Upload points ## Upload points
@@ -1424,7 +1424,7 @@ To delete entire points, see [deleting points](#delete-points).
### Update payload ### Update payload
Learn how to modify the payload of a point in the [Payload](../payload/#update-payload) section. Learn how to modify the payload of a point in the [Payload](/documentation/concepts/payload/#update-payload) section.
## Delete points ## Delete points
@@ -1716,7 +1716,7 @@ Python client:
Sometimes it might be necessary to get all stored points without knowing ids, or iterate over points that correspond to a filter. Sometimes it might be necessary to get all stored points without knowing ids, or iterate over points that correspond to a filter.
REST API ([Schema](https://api.qdrant.tech/master/api-reference/search/scroll-points)): REST API ([Schema](https://api.qdrant.tech/master/api-reference/points/scroll-points)):
```http ```http
POST /collections/{collection_name}/points/scroll POST /collections/{collection_name}/points/scroll
@@ -26,13 +26,13 @@ Depending on the `query` parameter, Qdrant might prefer different strategies for
| --- | --- | | --- | --- |
| Nearest Neighbors Search | Vector Similarity Search, also known as k-NN | | Nearest Neighbors Search | Vector Similarity Search, also known as k-NN |
| Search By Id | Search by an already stored vector - skip embedding model inference | | Search By Id | Search by an already stored vector - skip embedding model inference |
| [Recommendations](../explore/#recommendation-api) | Provide positive and negative examples | | [Recommendations](/documentation/concepts/explore/#recommendation-api) | Provide positive and negative examples |
| [Discovery Search](../explore/#discovery-api) | Guide the search using context as a one-shot training set | | [Discovery Search](/documentation/concepts/explore/#discovery-api) | Guide the search using context as a one-shot training set |
| [Scroll](../points/#scroll-points) | Get all points with optional filtering | | [Scroll](/documentation/concepts/points/#scroll-points) | Get all points with optional filtering |
| [Grouping](../search/#grouping-api) | Group results by a certain field | | [Grouping](/documentation/concepts/search/#grouping-api) | Group results by a certain field |
| [Order By](../hybrid-queries/#re-ranking-with-stored-values) | Order points by payload key | | [Order By](/documentation/concepts/hybrid-queries/#re-ranking-with-stored-values) | Order points by payload key |
| [Hybrid Search](../hybrid-queries/#hybrid-search) | Combine multiple queries to get better results | | [Hybrid Search](/documentation/concepts/hybrid-queries/#hybrid-search) | Combine multiple queries to get better results |
| [Multi-Stage Search](../hybrid-queries/#multi-stage-queries) | Optimize performance for large embeddings | | [Multi-Stage Search](/documentation/concepts/hybrid-queries/#multi-stage-queries) | Optimize performance for large embeddings |
| [Random Sampling](#random-sampling) | Get random points from the collection | | [Random Sampling](#random-sampling) | Get random points from the collection |
**Nearest Neighbors Search** **Nearest Neighbors Search**
@@ -406,7 +406,7 @@ Currently, it could be:
* `indexed_only` - With this option you can disable the search in those segments where vector index is not built yet. This may be useful if you want to minimize the impact to the search performance whilst the collection is also being updated. Using this option may lead to a partial result if the collection is not fully indexed yet, consider using it only if eventual consistency is acceptable for your use case. * `indexed_only` - With this option you can disable the search in those segments where vector index is not built yet. This may be useful if you want to minimize the impact to the search performance whilst the collection is also being updated. Using this option may lead to a partial result if the collection is not fully indexed yet, consider using it only if eventual consistency is acceptable for your use case.
Since the `filter` parameter is specified, the search is performed only among those points that satisfy the filter condition. Since the `filter` parameter is specified, the search is performed only among those points that satisfy the filter condition.
See details of possible filters and their work in the [filtering](../filtering/) section. See details of possible filters and their work in the [filtering](/documentation/concepts/filtering/) section.
Example result of this API would be Example result of this API would be
@@ -1282,7 +1282,7 @@ The result of this API contains one array per search requests.
*Available as of v0.8.3* *Available as of v0.8.3*
Search and [recommendation](../explore/#recommendation-api) APIs allow to skip first results of the search and return only the result starting from some specified offset: Search and [recommendation](/documentation/concepts/explore/#recommendation-api) APIs allow to skip first results of the search and return only the result starting from some specified offset:
Example: Example:
@@ -1423,7 +1423,7 @@ Using an `offset` parameter, will require to internally retrieve `offset + limit
It is possible to group results by a certain field. This is useful when you have multiple points for the same item, and you want to avoid redundancy of the same item in the results. It is possible to group results by a certain field. This is useful when you have multiple points for the same item, and you want to avoid redundancy of the same item in the results.
For example, if you have a large document split into multiple chunks, and you want to search or [recommend](../explore/#recommendation-api) on a per-document basis, you can group the results by the document ID. For example, if you have a large document split into multiple chunks, and you want to search or [recommend](/documentation/concepts/explore/#recommendation-api) on a per-document basis, you can group the results by the document ID.
Consider having points with the following payloads: Consider having points with the following payloads:
@@ -1631,7 +1631,7 @@ If the `group_by` field of a point is an array (e.g. `"document_id": ["a", "b"]`
**Limitations**: **Limitations**:
* Only [keyword](../payload/#keyword) and [integer](../payload/#integer) payload values are supported for the `group_by` parameter. Payload values with other types will be ignored. * Only [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payload values are supported for the `group_by` parameter. Payload values with other types will be ignored.
* At the moment, pagination is not enabled when using **groups**, so the `offset` parameter is not allowed. * At the moment, pagination is not enabled when using **groups**, so the `offset` parameter is not allowed.
### Lookup in groups ### Lookup in groups
@@ -1964,10 +1964,10 @@ This process is called query planning.
The strategy selection process relies heavily on heuristics and can vary from release to release. The strategy selection process relies heavily on heuristics and can vary from release to release.
However, the general principles are: However, the general principles are:
* planning is performed for each segment independently (see [storage](../storage/) for more information about segments) * planning is performed for each segment independently (see [storage](/documentation/concepts/storage/) for more information about segments)
* prefer a full scan if the amount of points is below a threshold * prefer a full scan if the amount of points is below a threshold
* estimate the cardinality of a filtered result before selecting a strategy * estimate the cardinality of a filtered result before selecting a strategy
* retrieve points using payload index (see [indexing](../indexing/)) if cardinality is below threshold * retrieve points using payload index (see [indexing](/documentation/concepts/indexing/)) if cardinality is below threshold
* use filterable vector index if the cardinality is above a threshold * use filterable vector index if the cardinality is above a threshold
You can adjust the threshold using a [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), as well as independently for each collection. You can adjust the threshold using a [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), as well as independently for each collection.
@@ -612,7 +612,7 @@ also configure to use an [S3 storage](#s3) service for them.
By default, snapshots are stored at `./snapshots` or at `/qdrant/snapshots` when By default, snapshots are stored at `./snapshots` or at `/qdrant/snapshots` when
using our Docker image. using our Docker image.
The target directory can be controlled through the [configuration](../../guides/configuration/): The target directory can be controlled through the [configuration](/documentation/guides/configuration/):
```yaml ```yaml
storage: storage:
@@ -640,7 +640,7 @@ storage:
Rather than storing snapshots on the local file system, you may also configure Rather than storing snapshots on the local file system, you may also configure
to store snapshots in an S3-compatible storage service. To enable this, you must to store snapshots in an S3-compatible storage service. To enable this, you must
configure it in the [configuration](../../guides/configuration/) file. configure it in the [configuration](/documentation/guides/configuration/) file.
For example, to configure for AWS S3: For example, to configure for AWS S3:
@@ -13,7 +13,7 @@ Each segment has its independent vector and payload storage as well as indexes.
Data stored in segments usually do not overlap. Data stored in segments usually do not overlap.
However, storing the same point in different segments will not cause problems since the search contains a deduplication mechanism. However, storing the same point in different segments will not cause problems since the search contains a deduplication mechanism.
The segments consist of vector and payload storages, vector and payload [indexes](../indexing/), and id mapper, which stores the relationship between internal and external ids. The segments consist of vector and payload storages, vector and payload [indexes](/documentation/concepts/indexing/), and id mapper, which stores the relationship between internal and external ids.
A segment can be `appendable` or `non-appendable` depending on the type of storage and index used. A segment can be `appendable` or `non-appendable` depending on the type of storage and index used.
You can freely add, delete and query data in the `appendable` segment. You can freely add, delete and query data in the `appendable` segment.
@@ -161,8 +161,8 @@ This is the recommended way, in case your Qdrant instance operates with fast dis
There are two ways to do this: There are two ways to do this:
1. You can set the threshold globally in the [configuration file](../../guides/configuration/). The parameter is called `memmap_threshold_kb`. 1. You can set the threshold globally in the [configuration file](/documentation/guides/configuration/). The parameter is called `memmap_threshold_kb`.
2. You can set the threshold for each collection separately during [creation](../collections/#create-collection) or [update](../collections/#update-collection-parameters). 2. You can set the threshold for each collection separately during [creation](/documentation/concepts/collections/#create-collection) or [update](/documentation/concepts/collections/#update-collection-parameters).
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -295,7 +295,7 @@ The rule of thumb to set the memmap threshold parameter is simple:
- if you have a high write load and low RAM - set memmap threshold lower than `indexing_threshold` to e.g. 10000. In this case the optimizer will convert the segments to memmap storage first and will only apply indexing after that. - if you have a high write load and low RAM - set memmap threshold lower than `indexing_threshold` to e.g. 10000. In this case the optimizer will convert the segments to memmap storage first and will only apply indexing after that.
In addition, you can use memmap storage not only for vectors, but also for HNSW index. In addition, you can use memmap storage not only for vectors, but also for HNSW index.
To enable this, you need to set the `hnsw_config.on_disk` parameter to `true` during collection [creation](../collections/#create-a-collection) or [updating](../collections/#update-collection-parameters). To enable this, you need to set the `hnsw_config.on_disk` parameter to `true` during collection [creation](/documentation/concepts/collections/#create-a-collection) or [updating](/documentation/concepts/collections/#update-collection-parameters).
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -452,7 +452,7 @@ If you need to query vectors with some payload-based conditions - checking value
In this scenario, we recommend creating a payload index for each field used in filtering conditions to avoid disk access. In this scenario, we recommend creating a payload index for each field used in filtering conditions to avoid disk access.
Once you create the field index, Qdrant will preserve all values of the indexed field in RAM regardless of the payload storage type. Once you create the field index, Qdrant will preserve all values of the indexed field in RAM regardless of the payload storage type.
You can specify the desired type of payload storage with [configuration file](../../guides/configuration/) or with collection parameter `on_disk_payload` during [creation](../collections/#create-collection) of the collection. You can specify the desired type of payload storage with [configuration file](/documentation/guides/configuration/) or with collection parameter `on_disk_payload` during [creation](/documentation/concepts/collections/#create-collection) of the collection.
## Versioning ## Versioning
@@ -7,15 +7,14 @@ weight: 18
| Integration | Description | | Integration | Description |
| ------------------------------- | -------------------------------------------------------------------------------------------------- | | ------------------------------- | -------------------------------------------------------------------------------------------------- |
| [Airbyte](./airbyte/) | Data integration platform specialising in ELT pipelines. | | [Airbyte](/documentation/data-management/airbyte/) | Data integration platform specialising in ELT pipelines. |
| [Airflow](./airflow/) | Platform designed for developing, scheduling, and monitoring batch-oriented workflows. | | [Airflow](/documentation/data-management/airflow/) | Platform designed for developing, scheduling, and monitoring batch-oriented workflows. |
| [Connect](./redpanda/) | Declarative data-agnostic streaming service for efficient, stateless processing. | | [Connect](/documentation/data-management/redpanda/) | Declarative data-agnostic streaming service for efficient, stateless processing. |
| [Confluent](./confluent/) | Fully-managed data streaming platform with a cloud-native Apache Kafka engine. | | [Confluent](/documentation/data-management/confluent/) | Fully-managed data streaming platform with a cloud-native Apache Kafka engine. |
| [DLT](./dlt/) | Python library to simplify data loading processes between several sources and destinations. | | [DLT](/documentation/data-management/dlt/) | Python library to simplify data loading processes between several sources and destinations. |
| [Fluvio](./fluvio/) | Rust-based platform for high speed, real-time data processing. | | [Fluvio](/documentation/data-management/fluvio/) | Rust-based platform for high speed, real-time data processing. |
| [Fondant](./fondant/) | Framework for developing datasets, sharing reusable operations and data processing trees. | | [Fondant](/documentation/data-management/fondant/) | Framework for developing datasets, sharing reusable operations and data processing trees. |
| [MindsDB](./mindsdb/) | Platform to deploy, serve, and fine-tune models with numerous data source integrations. | | [MindsDB](/documentation/data-management/mindsdb/) | Platform to deploy, serve, and fine-tune models with numerous data source integrations. |
| [NiFi](./nifi/) | Data ingestion platform to manage data transfer between different sources and destination systems. | | [NiFi](/documentation/data-management/nifi/) | Data ingestion platform to manage data transfer between different sources and destination systems. |
| [Spark](./spark/) | A unified analytics engine for large-scale data processing. | | [Spark](/documentation/data-management/spark/) | A unified analytics engine for large-scale data processing. |
| [Unstructured](./unstructured/) | Python library with components for ingesting and pre-processing data from numerous sources. | | [Unstructured](/documentation/data-management/unstructured/) | Python library with components for ingesting and pre-processing data from numerous sources. |
@@ -17,19 +17,19 @@ Additionally, [any open-source embeddings from HuggingFace](https://huggingface.
| Embeddings Providers | Description | | Embeddings Providers | Description |
| ----------------------------- | ----------- | | ----------------------------- | ----------- |
| [Aleph Alpha](./aleph-alpha/) | Multilingual embeddings focused on European languages. | | [Aleph Alpha](/documentation/embeddings/aleph-alpha/) | Multilingual embeddings focused on European languages. |
| [Bedrock](./bedrock/) | AWS managed service for foundation models and embeddings. | | [Bedrock](/documentation/embeddings/bedrock/) | AWS managed service for foundation models and embeddings. |
| [Cohere](./cohere/) | Language model embeddings for NLP tasks. | | [Cohere](/documentation/embeddings/cohere/) | Language model embeddings for NLP tasks. |
| [Gemini](./gemini/) | Google’s multimodal embeddings for text and vision. | [Gemini](/documentation/embeddings/gemini/) | Google’s multimodal embeddings for text and vision.
| [Jina AI](./jina-embeddings/) | Customizable embeddings for neural search. | | [Jina AI](/documentation/embeddings/jina-embeddings/) | Customizable embeddings for neural search. |
| [Mistral](./mistral/) | Open-source, efficient language model embeddings. | | [Mistral](/documentation/embeddings/mistral/) | Open-source, efficient language model embeddings. |
| [MixedBread](./mixedbread/) | Lightweight embeddings for constrained environments. | | [MixedBread](/documentation/embeddings/mixedbread/) | Lightweight embeddings for constrained environments. |
| [Mixpeek](./mixpeek/) | Managed SDK for video chunking, embedding, and post-processing.​ | | [Mixpeek](/documentation/embeddings/mixpeek/) | Managed SDK for video chunking, embedding, and post-processing.​ |
| [Nomic](./nomic/) | Embeddings for data visualization. | | [Nomic](/documentation/embeddings/nomic/) | Embeddings for data visualization. |
| [Nvidia](./nvidia/) | GPU-optimized embeddings from Nvidia. | | [Nvidia](/documentation/embeddings/nvidia/) | GPU-optimized embeddings from Nvidia. |
| [Ollama](./ollama/) | Embeddings for conversational AI. | | [Ollama](/documentation/embeddings/ollama/) | Embeddings for conversational AI. |
| [OpenAI](./openai/) | Industry-leading embeddings for NLP. | | [OpenAI](/documentation/embeddings/openai/) | Industry-leading embeddings for NLP. |
| [Prem AI](./premai/) | Precise language embeddings. | | [Prem AI](/documentation/embeddings/premai/) | Precise language embeddings. |
| [Snowflake](./snowflake/) | Scalable embeddings for big data. | | [Snowflake](/documentation/embeddings/snowflake/) | Scalable embeddings for big data. |
| [Upstage](./upstage/) | Embeddings for speech and language tasks. | | [Upstage](/documentation/embeddings/upstage/) | Embeddings for speech and language tasks. |
| [Voyage AI](./voyage/) | Navigation and spatial understanding embeddings. | | [Voyage AI](/documentation/embeddings/voyage/) | Navigation and spatial understanding embeddings. |
@@ -6,16 +6,16 @@ weight: 26
| End-to-End Code Samples | Description | Stack | | End-to-End Code Samples | Description | Stack |
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------| |---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
| [Multitenancy with LlamaIndex](../examples/llama-index-multitenancy/) | Handle data coming from multiple users in LlamaIndex. | Qdrant, Python, LlamaIndex | | [Multitenancy with LlamaIndex](/documentation/examples/llama-index-multitenancy/) | Handle data coming from multiple users in LlamaIndex. | Qdrant, Python, LlamaIndex |
| [Implement custom connector for Cohere RAG](../examples/cohere-rag-connector/) | Bring data stored in Qdrant to Cohere RAG | Qdrant, Cohere, FastAPI | | [Implement custom connector for Cohere RAG](/documentation/examples/cohere-rag-connector/) | Bring data stored in Qdrant to Cohere RAG | Qdrant, Cohere, FastAPI |
| [Chatbot for Interactive Learning](../examples/rag-chatbot-red-hat-openshift-haystack/) | Build a Private RAG Chatbot for Interactive Learning | Qdrant, Haystack, OpenShift | | [Chatbot for Interactive Learning](/documentation/examples/rag-chatbot-red-hat-openshift-haystack/) | Build a Private RAG Chatbot for Interactive Learning | Qdrant, Haystack, OpenShift |
| [Information Extraction Engine](../examples/rag-chatbot-vultr-dspy-ollama/) | Build a Private RAG Information Extraction Engine | Qdrant, Vultr, DSPy, Ollama | | [Information Extraction Engine](/documentation/examples/rag-chatbot-vultr-dspy-ollama/) | Build a Private RAG Information Extraction Engine | Qdrant, Vultr, DSPy, Ollama |
| [System for Employee Onboarding](../examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/) | Build a RAG System for Employee Onboarding | Qdrant, Cohere, LangChain | | [System for Employee Onboarding](/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/) | Build a RAG System for Employee Onboarding | Qdrant, Cohere, LangChain |
| [System for Contract Management](../examples/rag-contract-management-stackit-aleph-alpha/) | Build a Region-Specific RAG System for Contract Management | Qdrant, Aleph Alpha, STACKIT | | [System for Contract Management](/documentation/examples/rag-contract-management-stackit-aleph-alpha/) | Build a Region-Specific RAG System for Contract Management | Qdrant, Aleph Alpha, STACKIT |
| [Question-Answering System for Customer Support](../examples/rag-customer-support-cohere-airbyte-aws/) | Build a RAG System for AI Customer Support | Qdrant, Cohere, Airbyte, AWS | | [Question-Answering System for Customer Support](/documentation/examples/rag-customer-support-cohere-airbyte-aws/) | Build a RAG System for AI Customer Support | Qdrant, Cohere, Airbyte, AWS |
| [Hybrid Search on PDF Documents](../examples/hybrid-search-llamaindex-jinaai/) | Develop a Hybrid Search System for Product PDF Manuals | Qdrant, LlamaIndex, Jina AI | [Hybrid Search on PDF Documents](/documentation/examples/hybrid-search-llamaindex-jinaai/) | Develop a Hybrid Search System for Product PDF Manuals | Qdrant, LlamaIndex, Jina AI
| [Blog-Reading RAG Chatbot](../examples/rag-chatbot-scaleway) | Develop a RAG-based Chatbot on Scaleway and with LangChain | Qdrant, LangChain, GPT-4o | [Blog-Reading RAG Chatbot](/documentation/examples/rag-chatbot-scaleway/) | Develop a RAG-based Chatbot on Scaleway and with LangChain | Qdrant, LangChain, GPT-4o
| [Movie Recommendation System](../examples/recommendation-system-ovhcloud/) | Build a Movie Recommendation System with LlamaIndex and With JinaAI | Qdrant | | [Movie Recommendation System](/documentation/examples/recommendation-system-ovhcloud/) | Build a Movie Recommendation System with LlamaIndex and With JinaAI | Qdrant |
## Notebooks ## Notebooks
@@ -28,7 +28,7 @@ Our Notebooks offer complex instructions that are supported with a throrough exp
| [Search and Recommend Newspaper Articles](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_text_data/qdrant_and_text_data.ipynb) | Work with text data to develop a semantic search and a recommendation engine for news articles. | Qdrant | | [Search and Recommend Newspaper Articles](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_text_data/qdrant_and_text_data.ipynb) | Work with text data to develop a semantic search and a recommendation engine for news articles. | Qdrant |
| [Recommendation System for Songs](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_audio_data/03_qdrant_101_audio.ipynb) | Use Qdrant to develop a music recommendation engine based on audio embeddings. | Qdrant | | [Recommendation System for Songs](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_audio_data/03_qdrant_101_audio.ipynb) | Use Qdrant to develop a music recommendation engine based on audio embeddings. | Qdrant |
| [Image Comparison System for Skin Conditions](https://colab.research.google.com/github/qdrant/examples/blob/master/qdrant_101_image_data/04_qdrant_101_cv.ipynb) | Use Qdrant to compare challenging images with labels representing different skin diseases. | Qdrant | | [Image Comparison System for Skin Conditions](https://colab.research.google.com/github/qdrant/examples/blob/master/qdrant_101_image_data/04_qdrant_101_cv.ipynb) | Use Qdrant to compare challenging images with labels representing different skin diseases. | Qdrant |
| [Question and Answer System with LlamaIndex](https://githubtocolab.com/qdrant/examples/blob/master/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb) | Combine Qdrant and LlamaIndex to create a self-updating Q&A system. | Qdrant, LlamaIndex, Cohere | | [Question and Answer System with LlamaIndex](https://github.com/qdrant/examples/blob/949669f001a03131afebf2ecd1e0ce63cab01c81/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb) | Combine Qdrant and LlamaIndex to create a self-updating Q&A system. | Qdrant, LlamaIndex, Cohere |
| [Extractive QA System](https://githubtocolab.com/qdrant/examples/blob/master/extractive_qa/extractive-question-answering.ipynb) | Extract answers directly from context to generate highly relevant answers. | Qdrant | | [Extractive QA System](https://githubtocolab.com/qdrant/examples/blob/master/extractive_qa/extractive-question-answering.ipynb) | Extract answers directly from context to generate highly relevant answers. | Qdrant |
| [Ecommerce Reverse Image Search](https://githubtocolab.com/qdrant/examples/blob/master/ecommerce_reverse_image_search/ecommerce-reverse-image-search.ipynb) | Accept images as search queries to receive semantically appropriate answers. | Qdrant | | [Ecommerce Reverse Image Search](https://githubtocolab.com/qdrant/examples/blob/master/ecommerce_reverse_image_search/ecommerce-reverse-image-search.ipynb) | Accept images as search queries to receive semantically appropriate answers. | Qdrant |
| [Basic RAG](https://githubtocolab.com/qdrant/examples/blob/master/rag-openai-qdrant/rag-openai-qdrant.ipynb) | Basic RAG pipeline with Qdrant and OpenAI SDKs. | OpenAI, Qdrant, FastEmbed | | [Basic RAG](https://githubtocolab.com/qdrant/examples/blob/master/rag-openai-qdrant/rag-openai-qdrant.ipynb) | Basic RAG pipeline with Qdrant and OpenAI SDKs. | OpenAI, Qdrant, FastEmbed |
@@ -63,7 +63,7 @@ Verify that mighty works by calling `curl https://<address>:5050/sentence-transf
} }
``` ```
For Qdrant, follow our [cloud documentation](../../cloud/cloud-quick-start/) to spin up a [free tier](https://cloud.qdrant.io/). Make sure to retrieve an API key. For Qdrant, follow our [cloud documentation](/documentation/cloud/cloud-quick-start/) to spin up a [free tier](https://cloud.qdrant.io/). Make sure to retrieve an API key.
## Implement model API ## Implement model API
@@ -234,7 +234,7 @@ llm = AlephAlpha(
Then, we can glue the components together and build the search process. `RetrievalQA` is a class that takes implements Then, we can glue the components together and build the search process. `RetrievalQA` is a class that takes implements
the Question Retrieval process, with a specified retriever and Large Language Model. The instance of `Qdrant` might be the Question Retrieval process, with a specified retriever and Large Language Model. The instance of `Qdrant` might be
converted into a retriever, with additional filter that will be passed to the `similarity_search` method. The filter converted into a retriever, with additional filter that will be passed to the `similarity_search` method. The filter
is created as [in a regular Qdrant query](../../../documentation/concepts/filtering/), with the `roles` field set to the is created as [in a regular Qdrant query](/documentation/concepts/filtering/), with the `roles` field set to the
user's roles. user's roles.
```python ```python
@@ -132,14 +132,14 @@ progress of the synchronization in the UI.
## RAG connector ## RAG connector
One of our previous tutorials, guides you step-by-step on [implementing custom connector for Cohere One of our previous tutorials, guides you step-by-step on [implementing custom connector for Cohere
RAG](../cohere-rag-connector/) with Cohere Embed v3 and Qdrant. You can just point it to use your Hybrid Cloud RAG](documentation/examples/cohere-rag-connector/) with Cohere Embed v3 and Qdrant. You can just point it to use your Hybrid Cloud
Qdrant instance running on AWS. Created connector might be deployed to Amazon Web Services in various ways, even in a Qdrant instance running on AWS. Created connector might be deployed to Amazon Web Services in various ways, even in a
[Serverless](https://aws.amazon.com/serverless/) manner using [AWS [Serverless](https://aws.amazon.com/serverless/) manner using [AWS
Lambda](https://aws.amazon.com/lambda/?c=ser&sec=srv). Lambda](https://aws.amazon.com/lambda/?c=ser&sec=srv).
In general, RAG connector has to expose a single endpoint that will accept POST requests with `query` parameter and In general, RAG connector has to expose a single endpoint that will accept POST requests with `query` parameter and
return the matching documents as JSON document with a specific structure. Our FastAPI implementation created [in the return the matching documents as JSON document with a specific structure. Our FastAPI implementation created [in the
related tutorial](../cohere-rag-connector/) is a perfect fit for this task. The only difference is that you related tutorial](documentation/examples/cohere-rag-connector/) is a perfect fit for this task. The only difference is that you
should point it to the Cohere models and Qdrant running on AWS infrastructure. should point it to the Cohere models and Qdrant running on AWS infrastructure.
> Our connector is a lightweight web service that exposes a single endpoint and glues the Cohere embedding model with > Our connector is a lightweight web service that exposes a single endpoint and glues the Cohere embedding model with
@@ -9,18 +9,18 @@ weight: 2
The primary source of memory usage is vector data. There are several ways to address that: The primary source of memory usage is vector data. There are several ways to address that:
- Configure [Quantization](../../guides/quantization/) to reduce the memory usage of vectors. - Configure [Quantization](/documentation/guides/quantization/) to reduce the memory usage of vectors.
- Configure on-disk vector storage - Configure on-disk vector storage
The choice of the approach depends on your requirements. The choice of the approach depends on your requirements.
Read more about [configuring the optimal](../../tutorials/optimize/) use of Qdrant. Read more about [configuring the optimal](/documentation/tutorials/optimize/) use of Qdrant.
### How do you choose the machine configuration? ### How do you choose the machine configuration?
There are two main scenarios of Qdrant usage in terms of resource consumption: There are two main scenarios of Qdrant usage in terms of resource consumption:
- **Performance-optimized** -- when you need to serve vector search as fast (many) as possible. In this case, you need to have as much vector data in RAM as possible. Use our [calculator](https://cloud.qdrant.io/calculator) to estimate the required RAM. - **Performance-optimized** -- when you need to serve vector search as fast (many) as possible. In this case, you need to have as much vector data in RAM as possible. Use our [calculator](https://cloud.qdrant.io/calculator) to estimate the required RAM.
- **Storage-optimized** -- when you need to store many vectors and minimize costs by compromising some search speed. In this case, pay attention to the disk speed instead. More about it in the article about [Memory Consumption](../../../articles/memory-consumption/). - **Storage-optimized** -- when you need to store many vectors and minimize costs by compromising some search speed. In this case, pay attention to the disk speed instead. More about it in the article about [Memory Consumption](/articles/memory-consumption/).
### I configured on-disk vector storage, but memory usage is still high. Why? ### I configured on-disk vector storage, but memory usage is still high. Why?
@@ -38,6 +38,6 @@ If you want to limit the memory usage of the service, we recommend using [limits
There are several possible reasons for that: There are several possible reasons for that:
- **Using filters without payload index** -- If you're performing a search with a filter but you don't have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](../../concepts/indexing/#payload-index). - **Using filters without payload index** -- If you're performing a search with a filter but you don't have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](/documentation/concepts/indexing/#payload-index).
- **Usage of on-disk vector storage with slow disks** -- If you're using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](../../../articles/memory-consumption/). - **Usage of on-disk vector storage with slow disks** -- If you're using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](/articles/memory-consumption/).
- **Large limit or non-optimal query parameters** -- A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues. - **Large limit or non-optimal query parameters** -- A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues.
@@ -53,7 +53,7 @@ If you're still seeing `"vector": null` in your results, it might be that the ve
### How can I search without a vector? ### How can I search without a vector?
You are likely looking for the [scroll](../../concepts/points/#scroll-points) method. It allows you to retrieve the records based on filters or even iterate over all the records in the collection. You are likely looking for the [scroll](/documentation/concepts/points/#scroll-points) method. It allows you to retrieve the records based on filters or even iterate over all the records in the collection.
### Does Qdrant support a full-text search or a hybrid search? ### Does Qdrant support a full-text search or a hybrid search?
@@ -64,10 +64,10 @@ What Qdrant can do:
- Search with full-text filters - Search with full-text filters
- Apply full-text filters to the vector search (i.e., perform vector search among the records with specific words or phrases) - Apply full-text filters to the vector search (i.e., perform vector search among the records with specific words or phrases)
- Do prefix search and semantic [search-as-you-type](../../../articles/search-as-you-type/) - Do prefix search and semantic [search-as-you-type](/articles/search-as-you-type/)
- Sparse vectors, as used in [SPLADE](https://github.com/naver/splade) or similar models - Sparse vectors, as used in [SPLADE](https://github.com/naver/splade) or similar models
- [Multi-vectors](../../concepts/vectors/#multivectors), for example ColBERT and other late-interaction models - [Multi-vectors](/documentation/concepts/vectors/#multivectors), for example ColBERT and other late-interaction models
- Combination of the [multiple searches](../../concepts/hybrid-queries/) - Combination of the [multiple searches](/documentation/concepts/hybrid-queries/)
What Qdrant doesn't plan to support: What Qdrant doesn't plan to support:
@@ -76,7 +76,7 @@ What Qdrant doesn't plan to support:
- Query analyzers and other NLP tools - Query analyzers and other NLP tools
Of course, you can always combine Qdrant with any specialized tool you need, including full-text search engines. Of course, you can always combine Qdrant with any specialized tool you need, including full-text search engines.
Read more about [our approach](../../../articles/hybrid-search/) to hybrid search. Read more about [our approach](/articles/hybrid-search/) to hybrid search.
## Collections ## Collections
@@ -87,11 +87,11 @@ It is _highly_ recommended not to create many small collections, as it will lead
We consider creating a collection for each user/dialog/document as an antipattern. We consider creating a collection for each user/dialog/document as an antipattern.
Please read more about collections, isolation, and multiple users in our [Multitenancy](../../tutorials/multiple-partitions/) tutorial. Please read more about collections, isolation, and multiple users in our [Multitenancy](/documentation/tutorials/multiple-partitions/) tutorial.
### How do I upload a large number of vectors into a Qdrant collection? ### How do I upload a large number of vectors into a Qdrant collection?
Read about our recommendations in the [bulk upload](../../tutorials/bulk-upload/) tutorial. Read about our recommendations in the [bulk upload](/documentation/tutorials/bulk-upload/) tutorial.
### Can I only store quantized vectors and discard full precision vectors? ### Can I only store quantized vectors and discard full precision vectors?
@@ -14,7 +14,7 @@ FastEmbed easily integrates with Qdrant for a variety of multimodal search purpo
|Beginner|Advanced| |Beginner|Advanced|
|:-:|:-:| |:-:|:-:|
|[Generate Text Embedings with FastEmbed](fastembed-quickstart/)|[Combine FastEmbed with Qdrant for Vector Search](fastembed-semantic-search/)| |[Generate Text Embedings with FastEmbed](/documentation/fastembed/fastembed-quickstart/)|[Combine FastEmbed with Qdrant for Vector Search](/documentation/fastembed/fastembed-semantic-search/)|
## Why is FastEmbed useful? ## Why is FastEmbed useful?
@@ -7,21 +7,22 @@ weight: 20
| Framework | Description | | Framework | Description |
| ------------------------------------- | ---------------------------------------------------------------------------------------------------- | | ------------------------------------- | ---------------------------------------------------------------------------------------------------- |
| [AutoGen](./autogen/) | Framework from Microsoft building LLM applications using multiple conversational agents. | | [AutoGen](/documentation/frameworks/autogen/) | Framework from Microsoft building LLM applications using multiple conversational agents. |
| [Canopy](./canopy/) | Framework from Pinecone for building RAG applications using LLMs and knowledge bases. | | [Canopy](/documentation/frameworks/canopy/) | Framework from Pinecone for building RAG applications using LLMs and knowledge bases. |
| [Cheshire Cat](./cheshire-cat/) | Framework to create personalized AI assistants using custom data. | | [Cheshire Cat](/documentation/frameworks/cheshire-cat/) | Framework to create personalized AI assistants using custom data. |
| [DocArray](./docarray/) | Python library for managing data in multi-modal AI applications. | | [DocArray](/documentation/frameworks/docarray/) | Python library for managing data in multi-modal AI applications. |
| [DSPy](./dspy/) | Framework for algorithmically optimizing LM prompts and weights. | | [DSPy](/documentation/frameworks/dspy/) | Framework for algorithmically optimizing LM prompts and weights. |
| [Fifty-One](./fifty-one/) | Toolkit for building high-quality datasets and computer vision models. | | [Fifty-One](/documentation/frameworks/fifty-one/) | Toolkit for building high-quality datasets and computer vision models. |
| [Genkit](./genkit/) | Framework to build, deploy, and monitor production-ready AI-powered apps. | | [Genkit](/documentation/frameworks/genkit/) | Framework to build, deploy, and monitor production-ready AI-powered apps. |
| [Haystack](./haystack/) | LLM orchestration framework to build customizable, production-ready LLM applications. | | [Haystack](/documentation/frameworks/haystack/) | LLM orchestration framework to build customizable, production-ready LLM applications. |
| [Langchain](./langchain/) | Python framework for building context-aware, reasoning applications using LLMs. | | [Langchain](/documentation/frameworks/langchain/) | Python framework for building context-aware, reasoning applications using LLMs. |
| [Langchain-Go](./langchain-go/) | Go framework for building context-aware, reasoning applications using LLMs. | | [Langchain-Go](/documentation/frameworks/langchain-go/) | Go framework for building context-aware, reasoning applications using LLMs. |
| [Langchain4j](./langchain4j/) | Java framework for building context-aware, reasoning applications using LLMs. | | [Langchain4j](/documentation/frameworks/langchain4j/) | Java framework for building context-aware, reasoning applications using LLMs. |
| [LlamaIndex](./llama-index/) | A data framework for building LLM applications with modular integrations. | | [LlamaIndex](/documentation/frameworks/llama-index/) | A data framework for building LLM applications with modular integrations. |
| [MemGPT](./memgpt/) | System to build LLM agents with long term memory & custom tools | | [Mem0](/documentation/frameworks/mem0/) | Self-improving memory layer for LLM applications, enabling personalized AI experiences. |
| [Pandas-AI](./pandas-ai/) | Python library to query/visualize your data (CSV, XLSX, PostgreSQL, etc.) in natural language | | [MemGPT](/documentation/frameworks/memgpt/) | System to build LLM agents with long term memory & custom tools |
| [Semantic Router](./semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. | | [Pandas-AI](/documentation/frameworks/pandas-ai/) | Python library to query/visualize your data (CSV, XLSX, PostgreSQL, etc.) in natural language |
| [Spring AI](./spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. | | [Semantic Router](/documentation/frameworks/semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. |
| [txtai](./txtai/) | Python library for semantic search, LLM orchestration and language model workflows. | | [Spring AI](/documentation/frameworks/spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. |
| [Vanna AI](./vanna-ai/) | Python RAG framework for SQL generation and querying. | | [txtai](/documentation/frameworks/txtai/) | Python library for semantic search, LLM orchestration and language model workflows. |
| [Vanna AI](/documentation/frameworks/vanna-ai/) | Python RAG framework for SQL generation and querying. |
@@ -25,9 +25,9 @@ CORE_PORT=1865
Cheshire Cat takes great advantage of the following features of Qdrant: Cheshire Cat takes great advantage of the following features of Qdrant:
* [Collection Aliases](../../concepts/collections/#collection-aliases) to manage the change from one embedder to another. * [Collection Aliases](/documentation/concepts/collections/#collection-aliases) to manage the change from one embedder to another.
* [Quantization](../../guides/quantization/) to obtain a good balance between speed, memory usage and quality of the results. * [Quantization](/documentation/guides/quantization/) to obtain a good balance between speed, memory usage and quality of the results.
* [Snapshots](../../concepts/snapshots/) to not miss any information. * [Snapshots](/documentation/concepts/snapshots/) to not miss any information.
* [Community](https://discord.com/invite/tdtYvXjC4h) * [Community](https://discord.com/invite/tdtYvXjC4h)
![RAG Pipeline](/documentation/frameworks/cheshire-cat/stregatto.jpg) ![RAG Pipeline](/documentation/frameworks/cheshire-cat/stregatto.jpg)
@@ -64,7 +64,7 @@ addition, there are a few optional parameters:
metadataPayloadKey: 'metadata'; metadataPayloadKey: 'metadata';
``` ```
- `collectionCreateOptions`: [Additional options](<(https://qdrant.tech/documentation/concepts/collections/#create-a-collection)>) when creating the Qdrant collection. - `collectionCreateOptions`: [Additional options](/documentation/concepts/collections/#create-a-collection/) when creating the Qdrant collection.
## Usage ## Usage
@@ -0,0 +1,66 @@
---
title: Mem0
---
![Mem0 Logo](/documentation/frameworks/mem0/mem0-banner.png)
[Mem0](https://mem0.ai) is a self-improving memory layer for LLM applications, enabling personalized AI experiences that save costs and delight users. Mem0 remembers user preferences, adapts to individual needs, and continuously improves over time, ideal for chatbots and AI systems.
Mem0 supports various vector store providers, including Qdrant, for efficient data handling and search capabilities.
## Installation
To install Mem0 with Qdrant support, use the following command:
```sh
pip install mem0ai
```
## Usage
Here's a basic example of how to use Mem0 with Qdrant:
```python
import os
from mem0 import Memory
os.environ["OPENAI_API_KEY"] = "sk-xx"
config = {
"vector_store": {
"provider": "qdrant",
"config": {
"collection_name": "test",
"host": "localhost",
"port": 6333,
}
}
}
m = Memory.from_config(config)
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
```
## Configuration
When configuring Mem0 to use Qdrant as the vector store, you can specify [various parameters](https://docs.mem0.ai/components/vectordbs/dbs/qdrant#config) in the `config` dictionary.
## Advanced Usage
Mem0 provides additional functionality for managing and querying your vector data. Here are some examples:
```python
# Search memories
related_memories = m.search(query="What are Alice's hobbies?", user_id="alice")
# Update existing memory
result = m.update(memory_id="m1", data="Likes to play tennis on weekends")
# Get memory history
history = m.history(memory_id="m1")
```
## Further Reading
- [Mem0 GitHub Repository](https://github.com/mem0ai/mem0)
- [Mem0 Documentation](https://docs.mem0.ai/)
@@ -44,7 +44,7 @@ example, by deleting a collection. After resolving Qdrant can be restarted
normally to continue operation. normally to continue operation.
In recovery mode, collection operations are limited to In recovery mode, collection operations are limited to
[deleting](../../concepts/collections/#delete-collection) a [deleting](/documentation/concepts/collections/#delete-collection) a
collection. That is because only collection metadata is loaded during recovery. collection. That is because only collection metadata is loaded during recovery.
To enable recovery mode with the Qdrant Docker image you must set the To enable recovery mode with the Qdrant Docker image you must set the
@@ -0,0 +1,109 @@
---
title: Capacity Planning
weight: 11
aliases:
- capacity
- /documentation/cloud/capacity-sizing
---
# Capacity Planning
When setting up your cluster, you'll need to figure out the right balance of **RAM** and **disk storage**. The best setup depends on a few things:
- How many vectors you have and their dimensions.
- The amount of payload data you're using and their indexes.
- What data you want to store in memory versus on disk.
- Your cluster's replication settings.
- Whether you're using quantization and how you’ve set it up.
## Calculating RAM size
You should store frequently accessed data in RAM for faster retrieval. If you want to keep all vectors in memory for optimal performance, you can use this rough formula for estimation:
```text
memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5
```
At the end, we multiply everything by 1.5. This extra 50% accounts for metadata (such as indexes and point versions) and temporary segments created during optimization.
Let's say you want to store 1 million vectors with 1024 dimensions:
```text
memory_size = 1,000,000 * 1024 * 4 bytes * 1.5
```
The memory_size is approximately 6,144,000,000 bytes, or about 5.72 GB.
Depending on the use case, large datasets can benefit from reduced memory requirements via [quantization](/documentation/guides/quantization/).
## Calculating payload size
This is always different. The size of the payload depends on the [structure and content of your data](/documentation/concepts/payload/#payload-types). For instance:
- **Text fields** consume space based on length and encoding (e.g. a large chunk of text vs a few words).
- **Floats** have fixed sizes of 8 bytes for `int64` or `float64`.
- **Boolean fields** typically consume 1 byte.
<aside role="alert">
The easiest way to calculate your payload size is to use a JSON size calculator.
</aside>
Calculating total payload size is similar to vectors. We have to multiply it by 1.5 for back-end indexing processes.
```text
total_payload_size = number_of_points * payload_size * 1.5
```
Let's say you want to store 1 million points with JSON payloads of 5KB:
```text
total_payload_size = 1,000,000 * 5KB * 1.5
```
The total_payload_size is approximately 5,000,000 bytes, or about 4.77 GB.
## Choosing disk over RAM
For optimal performance, you should store only frequently accessed data in RAM. The rest should be offloaded to the disk. For example, extra payload fields that you don't use for filtering can be stored on disk.
Only [indexed fields](/documentation/concepts/indexing/#payload-index) should be stored in RAM. You can read more about payload storage in the [Storage](/documentation/concepts/storage/#payload-storage) section.
### Storage-focused configuration
If your priority is to handle large volumes of vectors with average search latency, it's recommended to configure [memory-mapped (mmap) storage](/documentation/concepts/storage/#configuring-memmap-storage). In this setup, vectors are stored on disk in memory-mapped files, while only the most frequently accessed vectors are cached in RAM.
The amount of available RAM greatly impacts search performance. As a general rule, if you store half as many vectors in RAM, search latency will roughly double.
Disk speed is also crucial. [Contact us](/documentation/support/) if you have specific requirements for high-volume searches in our Cloud.
### Subgroup-oriented configuration
If your use case involves splitting vectors into multiple collections or subgroups based on payload values (e.g., serving searches for multiple users, each with their own subset of vectors), memory-mapped storage is recommended.
In this scenario, only the active subset of vectors will be cached in RAM, allowing for fast searches for the most recent and active users. You can estimate the required memory size as:
```text
memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5
```
Please refer to our [multitenancy](/documentation/guides/multiple-partitions/) documentation for more details on partitioning data in a Qdrant.
## Scaling disk space in Qdrant Cloud
Clusters supporting vector search require substantial disk space compared to other search systems. If you're running low on disk space, you can use the UI at [cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster.
<aside role="status">Note: If you increase disk space via the Qdrant UI, you cannot reduce it later.</aside>
When running low on disk space, consider the following benefits of scaling up:
- **Larger Datasets**: Supports larger datasets, which can improve the relevance and quality of search results.
- **Improved Indexing**: Enables the use of advanced indexing strategies like HNSW.
- **Caching**: Enhances speed by having more RAM, allowing more frequently accessed data to be cached.
- **Backups and Redundancy**: Facilitates more frequent backups, which is a key advantage for data safety.
Always remember to add 50% of the vector size. This would account for things like indexes and auxiliary data used during operations such as vector insertion, deletion, and search. Thus, the estimated memory size including metadata is:
```text
total_vector_size = number_of_dimensions * 4 bytes * 1.5
```
**Disclaimer**
The above calculations are estimates at best. If you're looking for more accurate numbers, you should always test your data set in practice.
@@ -42,7 +42,7 @@ Can't open Collections meta Wal: Os { code: 11, kind: WouldBlock, message: "Reso
``` ```
It means that Qdrant cannot start because a collection cannot be loaded. Its It means that Qdrant cannot start because a collection cannot be loaded. Its
associated [WAL](../../concepts/storage/#versioning) files are currently associated [WAL](/documentation/concepts/storage/#versioning) files are currently
unavailable, likely because the same files are already being used by another unavailable, likely because the same files are already being used by another
Qdrant instance. Qdrant instance.
@@ -18,7 +18,7 @@ production mode, you could also choose to overwrite `config/production.yaml`.
See [ordering](#order-and-priority) for details on how configurations are See [ordering](#order-and-priority) for details on how configurations are
loaded. loaded.
The [Installation](../installation/) guide contains examples of how to set up Qdrant with a custom configuration for the different deployment methods. The [Installation](/documentation/guides/installation/) guide contains examples of how to set up Qdrant with a custom configuration for the different deployment methods.
## Order and priority ## Order and priority
@@ -32,7 +32,7 @@ In summary, single-node clusters are best for non-production workloads, replicat
## Enabling distributed mode in self-hosted Qdrant ## Enabling distributed mode in self-hosted Qdrant
To enable distributed deployment - enable the cluster mode in the [configuration](../configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`. To enable distributed deployment - enable the cluster mode in the [configuration](/documentation/guides/configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`.
```yaml ```yaml
cluster: cluster:
@@ -152,7 +152,7 @@ Qdrant uses the [Raft](https://raft.github.io/) consensus protocol to maintain c
Operations on points, on the other hand, do not go through the consensus infrastructure. Operations on points, on the other hand, do not go through the consensus infrastructure.
Qdrant is not intended to have strong transaction guarantees, which allows it to perform point operations with low overhead. Qdrant is not intended to have strong transaction guarantees, which allows it to perform point operations with low overhead.
In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](../../concepts/points/#awaiting-result) to see the results of your writes. In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](/documentation/concepts/points/#awaiting-result) to see the results of your writes.
Operations on collections, on the contrary, are part of the consensus which guarantees that all operations are durable and eventually executed by all nodes. Operations on collections, on the contrary, are part of the consensus which guarantees that all operations are durable and eventually executed by all nodes.
In practice it means that a majority of nodes agree on what operations should be applied before the service will perform them. In practice it means that a majority of nodes agree on what operations should be applied before the service will perform them.
@@ -171,7 +171,7 @@ There are two methods of distributing points across shards:
- **User-defined sharding**: _Available as of v1.7.0_ - Each point is uploaded to a specific shard, so that operations can hit only the shard or shards they need. Even with this distribution, shards still ensure having non-intersecting subsets of points. [See more...](#user-defined-sharding) - **User-defined sharding**: _Available as of v1.7.0_ - Each point is uploaded to a specific shard, so that operations can hit only the shard or shards they need. Even with this distribution, shards still ensure having non-intersecting subsets of points. [See more...](#user-defined-sharding)
Each node knows where all parts of the collection are stored through the [consensus protocol](./#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result. Each node knows where all parts of the collection are stored through the [consensus protocol](#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result.
### Choosing the right number of shards ### Choosing the right number of shards
@@ -667,7 +667,7 @@ fastest depends on the size and state of a shard.
Available shard transfer methods are: Available shard transfer methods are:
- `stream_records`: _(default)_ transfer by streaming just its records to the target node in batches. - `stream_records`: _(default)_ transfer by streaming just its records to the target node in batches.
- `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](../../concepts/snapshots/) automatically. - `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](/documentation/concepts/snapshots/) automatically.
- `wal_delta`: _(auto recovery default)_ transfer by resolving [WAL] difference; the operations that were missed. - `wal_delta`: _(auto recovery default)_ transfer by resolving [WAL] difference; the operations that were missed.
Each has pros, cons and specific requirements, some of which are: Each has pros, cons and specific requirements, some of which are:
@@ -720,7 +720,7 @@ are acceptable in your use case. If your cluster is unstable and out of
resources, it's probably best to use the `stream_records` transfer method, resources, it's probably best to use the `stream_records` transfer method,
because it is unlikely to fail. because it is unlikely to fail.
The `snapshot` transfer method utilizes [snapshots](../../concepts/snapshots/) The `snapshot` transfer method utilizes [snapshots](/documentation/concepts/snapshots/)
to transfer a shard. A snapshot is created automatically. It is then transferred to transfer a shard. A snapshot is created automatically. It is then transferred
and restored on the target node. After this is done, the snapshot is removed and restored on the target node. After this is done, the snapshot is removed
from both nodes. While the snapshot/transfer/restore operation is happening, the from both nodes. While the snapshot/transfer/restore operation is happening, the
@@ -749,7 +749,7 @@ The `stream_records` method is currently used as default. This may change in the
future. As of Qdrant 1.9.0 `wal_delta` is used for automatic shard replications future. As of Qdrant 1.9.0 `wal_delta` is used for automatic shard replications
to recover dead shards. to recover dead shards.
[WAL]: ../../concepts/storage/#versioning [WAL]: /documentation/concepts/storage/#versioning
## Replication ## Replication
@@ -985,7 +985,7 @@ Snapshot recovery, used in single-node deployment, is different from cluster one
Consensus manages all metadata about all collections and does not require snapshots to recover it. Consensus manages all metadata about all collections and does not require snapshots to recover it.
But you can use snapshots to recover missing shards of the collections. But you can use snapshots to recover missing shards of the collections.
Use the [Collection Snapshot Recovery API](../../concepts/snapshots/#recover-in-cluster-deployment) to do it. Use the [Collection Snapshot Recovery API](/documentation/concepts/snapshots/#recover-in-cluster-deployment) to do it.
The service will download the specified snapshot of the collection and recover shards with data from it. The service will download the specified snapshot of the collection and recover shards with data from it.
Once all shards of the collection are recovered, the collection will become operational again. Once all shards of the collection are recovered, the collection will become operational again.
@@ -206,4 +206,4 @@ After a successful build, you can find the binary in the following subdirectory
## Client libraries ## Client libraries
In addition to the service, Qdrant provides a variety of client libraries for different programming languages. For a full list, see our [Client libraries](../../interfaces/#client-libraries) documentation. In addition to the service, Qdrant provides a variety of client libraries for different programming languages. For a full list, see our [Client libraries](/documentation/interfaces/#client-libraries) documentation.
@@ -75,7 +75,7 @@ Qdrant server.
These currently provide the most basic status response, returning HTTP 200 if These currently provide the most basic status response, returning HTTP 200 if
Qdrant is started and ready to be used. Qdrant is started and ready to be used.
Regardless of whether an [API key](../security/#authentication) is configured, Regardless of whether an [API key](/documentation/guides/security/#authentication) is configured,
the endpoints are always accessible. the endpoints are always accessible.
You can read more about Kubernetes health endpoints You can read more about Kubernetes health endpoints
@@ -1,26 +1,30 @@
--- ---
title: Optimize Resources title: Optimize Performance
weight: 11 weight: 11
aliases: aliases:
- ../tutorials/optimize - ../tutorials/optimize
--- ---
# Optimize Qdrant # Optimizing Qdrant Performance: Three Scenarios
Different use cases have different requirements for balancing between memory, speed, and precision. Different use cases require different balances between memory usage, search speed, and precision. Qdrant is designed to be flexible and customizable so you can tune it to your specific needs.
Qdrant is designed to be flexible and customizable so you can tune it to your needs.
![Trafeoff](/docs/tradeoff.png) This guide will walk you three main optimization strategies:
- High Speed Search & Low Memory Usage
- High Precision & Low Memory Usage
- High Precision & High Speed Search
Let's look deeper into each of those possible optimization scenarios. ![qdrant resource tradeoffs](/docs/tradeoff.png)
## Prefer low memory footprint with high speed search ## 1. High-Speed Search with Low Memory Usage
The main way to achieve high speed search with low memory footprint is to keep vectors on disk while at the same time minimizing the number of disk reads. To achieve high search speed with minimal memory usage, you can store vectors on disk while minimizing the number of disk reads. Vector quantization is a technique that compresses vectors, allowing more of them to be stored in memory, thus reducing the need to read from disk.
Vector quantization is one way to achieve this. Quantization converts vectors into a more compact representation, which can be stored in memory and used for search. With smaller vectors you can cache more in RAM and reduce the number of disk reads. To configure in-memory quantization, with on-disk original vectors, you need to create a collection with the following parameters:
To configure in-memory quantization, with on-disk original vectors, you need to create a collection with the following configuration: - `on_disk`: Stores original vectors on disk.
- `quantization_config`: Compresses quantized vectors to `int8` using the `scalar` method.
- `always_ram`: Keeps quantized vectors in RAM.
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -180,9 +184,9 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
}) })
``` ```
`on_disk` will ensure that vectors will be stored on disk, while `always_ram` will ensure that quantized vectors will be stored in RAM. ### Disable Rescoring for Faster Search (optional)
Optionally, you can disable rescoring with search `params`, which will reduce the number of disk reads even further, but potentially slightly decrease the precision. This is completely optional. Disabling rescoring with search `params` can further reduce the number of disk reads. Note that this might slightly decrease precision.
```http ```http
POST /collections/{collection_name}/points/query POST /collections/{collection_name}/points/query
@@ -313,9 +317,11 @@ client.Query(context.Background(), &qdrant.QueryPoints{
}) })
``` ```
## Prefer high precision with low memory footprint ## 2. High Precision with Low Memory Usage
In case you need high precision, but don't have enough RAM to store vectors in memory, you can enable on-disk vectors and HNSW index. If you require high precision but have limited RAM, you can store both vectors and the HNSW index on disk. This setup reduces memory usage while maintaining search precision.
To store the vectors `on_disk`, you need to configure both the vectors and the HNSW index:
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -446,7 +452,9 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
}) })
``` ```
In this scenario you can increase the precision of the search by increasing the `ef` and `m` parameters of the HNSW index, even with limited RAM. ### Improving Precision
Increase the `ef` and `m` parameters of the HNSW index to improve precision, even with limited RAM:
```json ```json
... ...
@@ -458,15 +466,14 @@ In this scenario you can increase the precision of the search by increasing the
... ...
``` ```
The disk IOPS is a critical factor in this scenario, it will determine how fast you can perform search. **Note:** The speed of this setup depends on the disk’s IOPS (Input/Output Operations Per Second).</br>
You can use [fio](https://gist.github.com/superboum/aaa45d305700a7873a8ebbab1abddf2b) to measure disk IOPS. You can use [fio](https://gist.github.com/superboum/aaa45d305700a7873a8ebbab1abddf2b) to measure disk IOPS.
## Prefer high precision with high speed search ## 3. High Precision with High-Speed Search
For high speed and high precision search it is critical to keep as much data in RAM as possible. For scenarios requiring both high speed and high precision, keep as much data in RAM as possible. Apply quantization with re-scoring for tunable accuracy.
By default, Qdrant follows this approach, but you can tune it to your needs.
It is possible to achieve high search speed and tunable accuracy by applying quantization with re-scoring. Here is how you can configure scalar quantization for a collection:
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -622,7 +629,13 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
}) })
``` ```
There are also some search-time parameters you can use to tune the search accuracy and speed: ### Fine-Tuning Search Parameters
You can adjust search parameters like `hnsw_ef` and `exact` to balance between speed and precision:
**Key Parameters:**
- `hnsw_ef`: Number of neighbors to visit during search (higher value = better accuracy, slower speed).
- `exact`: Set to `true` for exact search, which is slower but more accurate. You can use it to compare results of the search with different `hnsw_ef` values versus the ground truth.
```http ```http
POST /collections/{collection_name}/points/query POST /collections/{collection_name}/points/query
@@ -737,19 +750,20 @@ client.Query(context.Background(), &qdrant.QueryPoints{
}) })
``` ```
- `hnsw_ef` - controls the number of neighbors to visit during search. The higher the value, the more accurate and slower the search will be. Recommended range is 32-512. ## Balancing Latency and Throughput
- `exact` - if set to `true`, will perform exact search, which will be slower, but more accurate. You can use it to compare results of the search with different `hnsw_ef` values versus the ground truth.
## Latency vs Throughput When optimizing search performance, latency and throughput are two main metrics to consider:
- **Latency:** Time taken for a single request.
- **Throughput:** Number of requests handled per second.
- There are two main approaches to measure the speed of search: The following optimization approaches are not mutually exclusive, but in some cases it might be preferable to optimize for one or another.
- latency of the request - the time from the moment request is submitted to the moment a response is received
- throughput - the number of requests per second the system can handle
Those approaches are not mutually exclusive, but in some cases it might be preferable to optimize for one or another. ### Minimizing Latency
To prefer minimizing latency, you can set up Qdrant to use as many cores as possible for a single request\. To minimize latency, you can set up Qdrant to use as many cores as possible for a single request.
You can do this by setting the number of segments in the collection to be equal to the number of cores in the system. In this case, each segment will be processed in parallel, and the final result will be obtained faster. You can do this by setting the number of segments in the collection to be equal to the number of cores in the system.
In this case, each segment will be processed in parallel, and the final result will be obtained faster.
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -877,10 +891,13 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
}, },
}) })
``` ```
### Maximizing Throughput
To prefer throughput, you can set up Qdrant to use as many cores as possible for processing multiple requests in parallel. To maximize throughput, configure Qdrant to use as many cores as possible to process multiple requests in parallel.
To do that, you can configure qdrant to use minimal number of segments, which is usually 2.
Large segments benefit from the size of the index and overall smaller number of vector comparisons required to find the nearest neighbors. But at the same time require more time to build index. To do that, use fewer segments (usually 2) to handle more requests in parallel.
Large segments benefit from the size of the index and overall smaller number of vector comparisons required to find the nearest neighbors. However, they will require more time to build the HNSW index.
```http ```http
PUT /collections/{collection_name} PUT /collections/{collection_name}
@@ -1007,4 +1024,14 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
DefaultSegmentNumber: qdrant.PtrOf(uint64(2)), DefaultSegmentNumber: qdrant.PtrOf(uint64(2)),
}, },
}) })
``` ```
## Summary
By adjusting configurations like vector storage, quantization, and search parameters, you can optimize Qdrant for different use cases:
- **Low Memory + High Speed:** Use vector quantization.
- **High Precision + Low Memory:** Store vectors and HNSW index on disk.
- **High Precision + High Speed:** Keep data in RAM, use quantization with re-scoring.
- **Latency vs. Throughput:** Adjust segment numbers based on the priority.
Choose the strategy that best fits your use case to get the most out of Qdrant’s performance capabilities.
@@ -7,6 +7,6 @@ weight: 21
| Integration | Description | | Integration | Description |
| ----------------------------------- | ------------------------------------------------------------------------------------------- | | ----------------------------------- | ------------------------------------------------------------------------------------------- |
| [Pulumi](./pulumi/) | Infrastructure as code tool for creating, deploying, and managing cloud infrastructure | | [Pulumi](/documentation/infrastructure/pulumi/) | Infrastructure as code tool for creating, deploying, and managing cloud infrastructure |
| [Terraform](./terraform/) | infrastructure as code tool to define resources in human-readable configuration files. | | [Terraform](/documentation/infrastructure/terraform/) | infrastructure as code tool to define resources in human-readable configuration files. |
| [Testcontainers](./testcontainers/) | Open source framework for providing throwaway, lightweight instances of systems for testing | | [Testcontainers](/documentation/infrastructure/testcontainers/) | Open source framework for providing throwaway, lightweight instances of systems for testing |
@@ -1,5 +1,6 @@
--- ---
title: Pulumi title: Pulumi
aliases: [ ../platforms/pulumi ]
--- ---
![Pulumi Logo](/documentation/platforms/pulumi/pulumi-logo.png) ![Pulumi Logo](/documentation/platforms/pulumi/pulumi-logo.png)
@@ -1,5 +1,6 @@
--- ---
title: Terraform title: Terraform
aliases: [ ../platforms/terraform ]
--- ---
![Terraform Logo](/documentation/platforms/terraform/terraform.png) ![Terraform Logo](/documentation/platforms/terraform/terraform.png)
@@ -7,6 +7,6 @@ weight: 22
| Tool | Description | | Tool | Description |
| ----------------------------- | -------------------------------------------------------------------------------------- | | ----------------------------- | -------------------------------------------------------------------------------------- |
| [OpenLIT](./openlit/) | Platform for OpenTelemetry-native Observability & Evals for LLMs and Vector Databases. | | [OpenLIT](/documentation/observability/openlit/) | Platform for OpenTelemetry-native Observability & Evals for LLMs and Vector Databases. |
| [OpenLLMetry](./openllmetry/) | Set of OpenTelemetry extensions to add Observability for your LLM application. | | [OpenLLMetry](/documentation/observability/openllmetry/) | Set of OpenTelemetry extensions to add Observability for your LLM application. |
| [Datadog](./datadog/) | Cloud-based monitoring and analytics platform. | | [Datadog](/documentation/observability/datadog/) | Cloud-based monitoring and analytics platform. |
@@ -108,17 +108,17 @@ Let's now evaluate, at a high-level, the way Qdrant is architected.
The diagram above represents a high-level overview of some of the main components of Qdrant. Here The diagram above represents a high-level overview of some of the main components of Qdrant. Here
are the terminologies you should get familiar with. are the terminologies you should get familiar with.
- [Collections](../concepts/collections/): A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](../concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements. - [Collections](/documentation/concepts/collections/): A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](/documentation/concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements.
- [Distance Metrics](https://en.wikipedia.org/wiki/Metric_space): These are used to measure - [Distance Metrics](https://en.wikipedia.org/wiki/Metric_space): These are used to measure
similarities among vectors and they must be selected at the same time you are creating a similarities among vectors and they must be selected at the same time you are creating a
collection. The choice of metric depends on the way the vectors were obtained and, in particular, collection. The choice of metric depends on the way the vectors were obtained and, in particular,
on the neural network that will be used to encode new queries. on the neural network that will be used to encode new queries.
- [Points](../concepts/points/): The points are the central entity that - [Points](/documentation/concepts/points/): The points are the central entity that
Qdrant operates with and they consist of a vector and an optional id and payload. Qdrant operates with and they consist of a vector and an optional id and payload.
- id: a unique identifier for your vectors. - id: a unique identifier for your vectors.
- Vector: a high-dimensional representation of data, for example, an image, a sound, a document, a video, etc. - Vector: a high-dimensional representation of data, for example, an image, a sound, a document, a video, etc.
- [Payload](../concepts/payload/): A payload is a JSON object with additional data you can add to a vector. - [Payload](/documentation/concepts/payload/): A payload is a JSON object with additional data you can add to a vector.
- [Storage](../concepts/storage/): Qdrant can use one of two options for - [Storage](/documentation/concepts/storage/): Qdrant can use one of two options for
storage, **In-memory** storage (Stores all vectors in RAM, has the highest speed since disk storage, **In-memory** storage (Stores all vectors in RAM, has the highest speed since disk
access is required only for persistence), or **Memmap** storage, (creates a virtual address access is required only for persistence), or **Memmap** storage, (creates a virtual address
space associated with the file on disk). space associated with the file on disk).
@@ -65,7 +65,7 @@ While doing a semantic search at scale, because this is what we sometimes call t
Vector search is an exciting alternative to sparse methods. It solves the issues we had with the keyword-based search without needing to maintain lots of heuristics manually. It requires an additional component, a neural encoder, to convert text into vectors. Vector search is an exciting alternative to sparse methods. It solves the issues we had with the keyword-based search without needing to maintain lots of heuristics manually. It requires an additional component, a neural encoder, to convert text into vectors.
[**Tutorial 1 - Qdrant for Complete Beginners**](/documentation/tutorials/search-beginners/) [**Tutorial 1 - Qdrant for Complete Beginners**](/documentation/tutorials/search-beginners/)
Despite its complicated background, vectors search is extraordinarily simple to set up. With Qdrant, you can have a search engine up-and-running in five minutes. Our [Complete Beginners tutorial](../../tutorials/search-beginners/) will show you how. Despite its complicated background, vectors search is extraordinarily simple to set up. With Qdrant, you can have a search engine up-and-running in five minutes. Our [Complete Beginners tutorial](/documentation/tutorials/search-beginners/) will show you how.
[**Tutorial 2 - Question and Answer System**](/articles/qa-with-cohere-and-qdrant/) [**Tutorial 2 - Question and Answer System**](/articles/qa-with-cohere-and-qdrant/)
However, you can also choose SaaS tools to generate them and avoid building your model. Setting up a vector search project with Qdrant Cloud and Cohere co.embed API is fairly easy if you follow the [Question and Answer system tutorial](/articles/qa-with-cohere-and-qdrant/). However, you can also choose SaaS tools to generate them and avoid building your model. Setting up a vector search project with Qdrant Cloud and Cohere co.embed API is fairly easy if you follow the [Question and Answer system tutorial](/articles/qa-with-cohere-and-qdrant/).
@@ -7,13 +7,13 @@ weight: 23
| Platform | Description | | Platform | Description |
| --------------------------- | ---------------------------------------------------------------------------------------- | | --------------------------- | ---------------------------------------------------------------------------------------- |
| [Apify](./apify/) | Platform to build web scrapers and automate web browser tasks. | | [Apify](/documentation/platforms/apify/) | Platform to build web scrapers and automate web browser tasks. |
| [Bubble](./bubble) | Development platform for application development with a no-code interface | | [Bubble](/documentation/platforms/bubble/) | Development platform for application development with a no-code interface |
| [BuildShip](./buildship) | Low-code visual builder to create APIs, scheduled jobs, and backend workflows. | | [BuildShip](/documentation/platforms/buildship/) | Low-code visual builder to create APIs, scheduled jobs, and backend workflows. |
| [DocsGPT](./docsgpt/) | Tool for ingesting documentation sources and enabling conversations and queries. | | [DocsGPT](/documentation/platforms/docsgpt/) | Tool for ingesting documentation sources and enabling conversations and queries. |
| [Make](./make/) | Cloud platform to build low-code workflows by integrating various software applications. | | [Make](/documentation/platforms/make/) | Cloud platform to build low-code workflows by integrating various software applications. |
| [N8N](./n8n/) | Platform for node-based, low-code workflow automation. | | [N8N](/documentation/platforms/n8n/) | Platform for node-based, low-code workflow automation. |
| [Pipedream](./pipedream/) | Platform for connecting apps and developing event-driven automation. | | [Pipedream](/documentation/platforms/pipedream/) | Platform for connecting apps and developing event-driven automation. |
| [Portable.io](./portable/) | Cloud platform for developing and deploying ELT transformations. | | [Portable.io](/documentation/platforms/portable/) | Cloud platform for developing and deploying ELT transformations. |
| [PrivateGPT](./privategpt/) | Tool to ask questions about your documents using local LLMs emphasising privacy. | | [PrivateGPT](/documentation/platforms/privategpt/) | Tool to ask questions about your documents using local LLMs emphasising privacy. |
| [Rivet](./rivet/) | A visual programming environment for building AI agents with LLMs. | | [Rivet](/documentation/platforms/rivet/) | A visual programming environment for building AI agents with LLMs. |
@@ -471,7 +471,7 @@ fmt.Println(searchResult)
``` ```
The results are returned in decreasing similarity order. Note that payload and vector data is missing in these results by default. The results are returned in decreasing similarity order. Note that payload and vector data is missing in these results by default.
See [payload and vector in the result](../concepts/search/#payload-and-vector-in-the-result) on how to enable it. See [payload and vector in the result](/documentation/concepts/search/#payload-and-vector-in-the-result) on how to enable it.
## Add a filter ## Add a filter
@@ -594,14 +594,14 @@ fmt.Println(searchResult)
] ]
``` ```
<aside role="status">To make filtered search fast on real datasets, we highly recommend to create <a href="../concepts/indexing/#payload-index">payload indexes</a>!</aside> <aside role="status">To make filtered search fast on real datasets, we highly recommend to create <a href="/documentation/concepts/indexing/#payload-index">payload indexes</a>!</aside>
You have just conducted vector search. You loaded vectors into a database and queried the database with a vector of your own. Qdrant found the closest results and presented you with a similarity score. You have just conducted vector search. You loaded vectors into a database and queried the database with a vector of your own. Qdrant found the closest results and presented you with a similarity score.
## Next steps ## Next steps
Now you know how Qdrant works. Getting started with [Qdrant Cloud](../cloud/quickstart-cloud/) is just as easy. [Create an account](https://qdrant.to/cloud) and use our SaaS completely free. We will take care of infrastructure maintenance and software updates. Now you know how Qdrant works. Getting started with [Qdrant Cloud](/documentation/cloud/quickstart-cloud/) is just as easy. [Create an account](https://qdrant.to/cloud) and use our SaaS completely free. We will take care of infrastructure maintenance and software updates.
To move onto some more complex examples of vector search, read our [Tutorials](../tutorials/) and create your own app with the help of our [Examples](../examples/). To move onto some more complex examples of vector search, read our [Tutorials](/documentation/tutorials/) and create your own app with the help of our [Examples](/documentation/examples/).
**Note:** There is another way of running Qdrant locally. If you are a Python developer, we recommend that you try Local Mode in [Qdrant Client](https://github.com/qdrant/qdrant-client), as it only takes a few moments to get setup. **Note:** There is another way of running Qdrant locally. If you are a Python developer, we recommend that you try Local Mode in [Qdrant Client](https://github.com/qdrant/qdrant-client), as it only takes a few moments to get setup.
@@ -8,6 +8,6 @@ weight: 25
| Example | Description | Stack | | Example | Description | Stack |
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------| |---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
| [Pinecone to Qdrant Data Transfer](https://githubtocolab.com/qdrant/examples/blob/master/data-migration/from-pinecone-to-qdrant.ipynb) | Migrate your vector data from Pinecone to Qdrant. | Qdrant, Vector-io | | [Pinecone to Qdrant Data Transfer](https://githubtocolab.com/qdrant/examples/blob/master/data-migration/from-pinecone-to-qdrant.ipynb) | Migrate your vector data from Pinecone to Qdrant. | Qdrant, Vector-io |
| [Stream Data to Qdrant with Kafka](../send-data/data-streaming-kafka-qdrant/) | Use Confluent to Stream Data to Qdrant via Managed Kafka. | Qdrant, Kafka | | [Stream Data to Qdrant with Kafka](/documentation/send-data/data-streaming-kafka-qdrant/) | Use Confluent to Stream Data to Qdrant via Managed Kafka. | Qdrant, Kafka |
| [Qdrant on Databricks](../send-data/databricks/) | Learn how to use Qdrant on Databricks using the Spark connector | Qdrant, Databricks, Apache Spark | | [Qdrant on Databricks](/documentation/send-data/databricks/) | Learn how to use Qdrant on Databricks using the Spark connector | Qdrant, Databricks, Apache Spark |
| [Qdrant with Airflow and Astronomer](../send-data/qdrant-airflow-astronomer/) | Build a semantic querying system using Airflow and Astronomer | Qdrant, Airflow, Astronomer | | [Qdrant with Airflow and Astronomer](/documentation/send-data/qdrant-airflow-astronomer/) | Build a semantic querying system using Airflow and Astronomer | Qdrant, Airflow, Astronomer |
@@ -14,14 +14,14 @@ These tutorials demonstrate different ways you can build vector search into your
| Essential How-Tos | Description | Stack | | Essential How-Tos | Description | Stack |
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------| |---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
| [Semantic Search for Beginners](../tutorials/search-beginners/) | Create a simple search engine locally in minutes. | Qdrant | | [Semantic Search for Beginners](/documentation/tutorials/search-beginners/) | Create a simple search engine locally in minutes. | Qdrant |
| [Simple Neural Search](../tutorials/neural-search/) | Build and deploy a neural search that browses startup data. | Qdrant, BERT, FastAPI | | [Simple Neural Search](/documentation/tutorials/neural-search/) | Build and deploy a neural search that browses startup data. | Qdrant, BERT, FastAPI |
| [Neural Search with FastEmbed](../tutorials/neural-search-fastembed/) | Build and deploy a neural search with our FastEmbed library. | Qdrant | | [Neural Search with FastEmbed](/documentation/tutorials/neural-search-fastembed/) | Build and deploy a neural search with our FastEmbed library. | Qdrant |
| [Multimodal Search](../tutorials/multimodal-search-fastembed/) | Create a simple multimodal search. | Qdrant | | [Multimodal Search](/documentation/tutorials/multimodal-search-fastembed/) | Create a simple multimodal search. | Qdrant |
| [Bulk Upload Vectors](../tutorials/bulk-upload/) | Upload a large scale dataset. | Qdrant | | [Bulk Upload Vectors](/documentation/tutorials/bulk-upload/) | Upload a large scale dataset. | Qdrant |
| [Asynchronous API](../tutorials/async-api/) | Communicate with Qdrant server asynchronously with Python SDK. | Qdrant, Python | | [Asynchronous API](/documentation/tutorials/async-api/) | Communicate with Qdrant server asynchronously with Python SDK. | Qdrant, Python |
| [Create Dataset Snapshots](../tutorials/create-snapshot/) | Turn a dataset into a snapshot by exporting it from a collection. | Qdrant | | [Create Dataset Snapshots](/documentation/tutorials/create-snapshot/) | Turn a dataset into a snapshot by exporting it from a collection. | Qdrant |
| [Load HuggingFace Dataset](../tutorials/huggingface-datasets/) | Load a Hugging Face dataset to Qdrant | Qdrant, Python, datasets | | [Load HuggingFace Dataset](/documentation/tutorials/huggingface-datasets/) | Load a Hugging Face dataset to Qdrant | Qdrant, Python, datasets |
| [Measure Retrieval Quality](../tutorials/retrieval-quality/) | Measure and fine-tune the retrieval quality | Qdrant, Python, datasets | | [Measure Retrieval Quality](/documentation/tutorials/retrieval-quality/) | Measure and fine-tune the retrieval quality | Qdrant, Python, datasets |
| [Search Through Code](../tutorials/code-search/) | Implement semantic search application for code search tasks | Qdrant, Python, sentence-transformers, Jina | | [Search Through Code](/documentation/tutorials/code-search/) | Implement semantic search application for code search tasks | Qdrant, Python, sentence-transformers, Jina |
| [Setup Collaborative Filtering](../tutorials/collaborative-filtering/) | Implement a collaborative filtering system for recommendation engines | Qdrant| | [Setup Collaborative Filtering](/documentation/tutorials/collaborative-filtering/) | Implement a collaborative filtering system for recommendation engines | Qdrant|
@@ -101,23 +101,23 @@ client.updateCollection("{collection_name}", {
## Upload directly to disk ## Upload directly to disk
When the vectors you upload do not all fit in RAM, you likely want to use When the vectors you upload do not all fit in RAM, you likely want to use
[memmap](../../concepts/storage/#configuring-memmap-storage) [memmap](/documentation/concepts/storage/#configuring-memmap-storage)
support. support.
During collection During collection
[creation](../../concepts/collections/#create-collection), [creation](/documentation/concepts/collections/#create-collection),
memmaps may be enabled on a per-vector basis using the `on_disk` parameter. This memmaps may be enabled on a per-vector basis using the `on_disk` parameter. This
will store vector data directly on disk at all times. It is suitable for will store vector data directly on disk at all times. It is suitable for
ingesting a large amount of data, essential for the billion scale benchmark. ingesting a large amount of data, essential for the billion scale benchmark.
Using `memmap_threshold_kb` is not recommended in this case. It would require Using `memmap_threshold_kb` is not recommended in this case. It would require
the [optimizer](../../concepts/optimizer/) to constantly the [optimizer](/documentation/concepts/optimizer/) to constantly
transform in-memory segments into memmap segments on disk. This process is transform in-memory segments into memmap segments on disk. This process is
slower, and the optimizer can be a bottleneck when ingesting a large amount of slower, and the optimizer can be a bottleneck when ingesting a large amount of
data. data.
Read more about this in Read more about this in
[Configuring Memmap Storage](../../concepts/storage/#configuring-memmap-storage). [Configuring Memmap Storage](/documentation/concepts/storage/#configuring-memmap-storage).
## Parallel upload into multiple shards ## Parallel upload into multiple shards
@@ -240,7 +240,7 @@ The query has been narrowed down to one result from 2008.
## Next Steps ## Next Steps
Congratulations, you have just created your very first search engine! Trust us, the rest of Qdrant is not that complicated, either. For your next tutorial you should try building an actual [Neural Search Service with a complete API and a dataset](../../tutorials/neural-search/). Congratulations, you have just created your very first search engine! Trust us, the rest of Qdrant is not that complicated, either. For your next tutorial you should try building an actual [Neural Search Service with a complete API and a dataset](/documentation/tutorials/neural-search/).
## Return to the bash shell ## Return to the bash shell
@@ -96,6 +96,9 @@ menuItems:
- id: 2 - id: 2
name: Articles name: Articles
url: /articles/ url: /articles/
- id: 3
name: Startup Program
url: /qdrant-for-startups/
- title: Company - title: Company
items: items:
- id: 0 - id: 0
+4
View File
@@ -99,6 +99,10 @@ menuItems:
name: Articles name: Articles
icon: articles.svg icon: articles.svg
url: /articles/ url: /articles/
- id: subMenu-3-3
name: Startup Program
icon: qdrant-for-startups.svg
url: /qdrant-for-startups/
- id: menu-4 - id: menu-4
name: Company name: Company
subMenuItems: subMenuItems:
+1 -1
View File
@@ -8,7 +8,7 @@ homeButton:
link: / link: /
text: Go to Home text: Go to Home
supportButton: supportButton:
link: https://qdrant.io/discord link: https://discord.gg/qdrant
text: Get Support text: Get Support
sitemapExclude: True sitemapExclude: True
--- ---
+2 -2
View File
@@ -1,6 +1,6 @@
--- ---
stats: stats:
githubStars: 19.9k githubStars: 20.0k
discordMembers: 6.5k discordMembers: 6.6k
twitterFollowers: 7.5k twitterFollowers: 7.5k
--- ---
@@ -10,11 +10,11 @@ icon: <svg width="16" height="16" viewBox="0 0 16 16" fill="none"
7.86204 15.74L14.5287 7.07333C14.6834 6.87199 14.71 6.59999 14.598 6.37199Z" 7.86204 15.74L14.5287 7.07333C14.6834 6.87199 14.71 6.59999 14.598 6.37199Z"
fill="#8547FF"/></g><defs><clipPath id="clip0_770_2716"><rect width="16" fill="#8547FF"/></g><defs><clipPath id="clip0_770_2716"><rect width="16"
height="16" fill="white"/></clipPath></defs></svg> height="16" fill="white"/></clipPath></defs></svg>
text: "Webinar: Building Agents with LlamaIndex & Qdrant" text: "Guide: Best Practices in RAG Evaluation"
link: link:
text: Register now text: Read now
url: https://try.qdrant.tech/build-advanced-agents-with-llamaindex-and-qdrant?utm_source=website&utm_medium=homepage-banner&utm_campaign=sept-webinar-agents-llamaindex url: https://qdrant.tech/rag/rag-evaluation-guide/?utm_source=website&utm_medium=homepage-banner&utm_campaign=rag-eval-guide
start: 2024-09-07T15:28:00.000Z start: 2024-10-01T15:28:00.000Z
sitemapExclude: true sitemapExclude: true
end: 2024-09-26T08:00:00.000Z end: 2024-10-31T08:00:00.000Z
--- ---
@@ -1,13 +1,11 @@
--- ---
title: Qdrant For Startups title: Qdrant for Startups
description: Qdrant For Startups description: Supporting early-stage startups with discounts on Qdrant Cloud, technical guidance, and access to key AI tools from LlamaIndex, Hugging Face, and Airbyte.
build:
render: always
cascade: cascade:
- _target: - build:
environment: production list: local
build: publishResources: false
list: never render: never
render: never
publishResources: false
sitemapExclude: true
# todo: remove sitemapExclude and change building options after the page is ready to be published
--- ---
@@ -2,7 +2,7 @@
title: Why join Qdrant for Startups? title: Why join Qdrant for Startups?
mainCard: mainCard:
title: Discount for Qdrant Cloud title: Discount for Qdrant Cloud
description: Receive a discount on <a href="https://cloud.qdrant.io/" target="_blank">Qdrant Cloud</a> for the first year. description: Enjoy a discount on <a href="https://cloud.qdrant.io/" target="_blank">Qdrant Cloud</a> for the first year.
image: image:
src: /img/qdrant-for-startups-benefits/card1.png src: /img/qdrant-for-startups-benefits/card1.png
alt: Qdrant Discount for Startups alt: Qdrant Discount for Startups
@@ -14,14 +14,13 @@ cards:
src: /img/qdrant-for-startups-benefits/card2.svg src: /img/qdrant-for-startups-benefits/card2.svg
alt: Expert Technical Advice alt: Expert Technical Advice
- id: 1 - id: 1
title: Co-Marketing Opportunities title: Partner Perks
description: We’d love to share your work with our community. Exclusive access to our Vector Space Talks, joint blog posts, and more. description: Receive exclusive perks from Hugging Face, LlamaIndex, and Airbyte, ensuring you have access to key tools and resources for AI-driven applications.
image: image:
src: /img/qdrant-for-startups-benefits/card3.svg src: /img/qdrant-for-startups-benefits/card3.svg
alt: Co-Marketing Opportunities alt: Co-Marketing Opportunities
description: Qdrant is the leading open source vector database and similarity search engine designed to handle high-dimensional vectors for performance and massive-scale AI applications. button:
link: text: Apply Now
url: /documentation/overview/ url: "#form"
text: Learn More
sitemapExclude: true sitemapExclude: true
--- ---
@@ -4,37 +4,38 @@ questions:
- id: 0 - id: 0
question: What are the eligibility requirements? question: What are the eligibility requirements?
answer: | answer: |
<p>You must meet all of the following:</p>
<ul> <ul>
<li>Pre-seed, Seed or Series A startups (under five years old)</li> <li>New user of Qdrant Cloud.</li>
<li>New user of Qdrant Cloud</li> <li>Pre-seed, Seed, or Series A startups (under five years old) and less than $5M in funding.</li>
<li>Not a previous participant in the Qdrant for Startups program</li> <li>Have not previously participated in the Qdrant for Startups program</li>
<li>Be building an AI-driven product or service (agencies or devshops are not eligible)</li> <li>Building an AI-driven product or services (agencies or devshops are not eligible)</li>
<li>Have a live, functional website</li> <li>Provide a link to a live, functional website</li>
<li>Billing must be done directly with Qdrant (not through a marketplace)</li> <li>Billing will be done directly with Qdrant (not through a marketplace)</li>
</ul> </ul>
- id: 1 - id: 1
question: When will I get notified about my application? question: How can I apply to the Qdrant Startup Program?
answer: Upon submitting your application, we will review it and notify you of your status within 7 business days. answer: Apply through our online form by providing details about your startup and plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities.
- id: 2 - id: 2
question: What is the price?
answer: It is free to apply to the program. As part of the program, you will receive a discount on Qdrant Cloud, valid for 12 months. For detailed cloud pricing, please visit qdrant.tech/pricing.
- id: 3
question: How can my startup join the program?
answer: Your startup can join the program by simply submitting the application on this page. Once submitted, we will review your application and notify you of your status within 7 business days.
- id: 4
question: What criteria are used to select startups for the program? question: What criteria are used to select startups for the program?
answer: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected. answer: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected.
- id: 3
question: How long is the discount valid, and what are the conditions?
answer: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount. Billing can not be through a marketplace. For details on pricing, please visit qdrant.tech/pricing.
- id: 4
question: How can I maximize the co-marketing opportunities offered by the program?
answer: Engage actively with our marketing team for features on social media, possible appearances in Discord talks or webinars, and case studies to maximize your startup's visibility and showcase your innovative use of Qdrant.
- id: 5 - id: 5
question: How long is the discount valid, and are there any conditions?
answer: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount.
- id: 6
question: Can existing Qdrant customers apply for the Startup Program? question: Can existing Qdrant customers apply for the Startup Program?
answer: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered by the program. answer: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered.
- id: 7 - id: 6
question: Can I reapply if my application is initially rejected? question: Can I reapply if my application is initially rejected?
answer: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply. answer: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply.
- id: 8 - id: 7
question: Who can I contact for more information about the program? question: Who can I contact for more information about the program?
answer: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com. answer: After reading these FAQs in full, if you need more details or assistance, please contact <a href="mailto:startups@qdrant.com">startups@qdrant.com</a>.
button:
text: Apply Now
url: "#form"
sitemapExclude: true sitemapExclude: true
--- ---
@@ -1,12 +1,11 @@
--- ---
title: Qdrant for Startups title: Qdrant for Startups
description: Powering The Next Wave of AI Innovators, Qdrant for Startups is committed to being the catalyst for the next generation of AI pioneers. Our program is specifically designed to provide AI-focused startups with the right resources to scale. If AI is at the heart of your startup, you're in the right place. description: Powering The Next Wave of AI Innovators, Qdrant for Startups is committed to being the catalyst for the next generation of AI pioneers.<br><br>Our program is specifically designed to provide AI-focused startups with the right resources to scale. If AI is at the heart of your startup, you're in the right place.
button: button:
text: Apply Now text: Apply Now
url: "#form" url: "#form"
image: image:
src: /img/qdrant-for-startups-hero.svg src: /img/startups-program.svg
srcMobile: /img/mobile/qdrant-for-startups-hero.svg
alt: Qdrant for Startups alt: Qdrant for Startups
sitemapExclude: true sitemapExclude: true
--- ---
@@ -0,0 +1,15 @@
---
label: A comprehensive guide
icon:
src: /icons/outline/guidebook-blue.svg
alt: Guidebook
title: Best Practices in RAG Evaluation
description: Learn how to assess, calibrate, and optimize your RAG applications for long-term success.
getGuideButton:
text: Get the Guide
url: "/rag/rag-evaluation-guide/"
image:
src: /img/retrieval-augmented-generation-evaluation/guide-graphic.svg
alt: RAG guide
sitemapExclude: true
---
@@ -9,8 +9,8 @@ features:
title: Question and Answer System with LlamaIndex title: Question and Answer System with LlamaIndex
description: Combine Qdrant and LlamaIndex to create a self-updating Q&A system. description: Combine Qdrant and LlamaIndex to create a self-updating Q&A system.
link: link:
text: Video Tutorial text: Jupyter Notebook
url: https://www.youtube.com/watch?v=id5ql-Abq4Y&t=56s url: https://github.com/qdrant/examples/blob/949669f001a03131afebf2ecd1e0ce63cab01c81/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb
- id: 1 - id: 1
image: image:
src: /img/retrieval-augmented-generation-use-cases/case2.svg src: /img/retrieval-augmented-generation-use-cases/case2.svg
@@ -1,4 +1,4 @@
<h2>Confirm your signup</h2> <h2>Confirm your signup</h2>
<p>Follow this link to confirm your user:</p> <p>Follow this link to confirm your user:</p>
<p><a href="{{ .SiteURL }}/admin/#confirmation_token={{ .Token }}">Confirm your mail</a></p> <p><a href="http://{{ .SiteURL }}/admin/#confirmation_token={{ .Token }}">Confirm your mail</a></p>
@@ -4,4 +4,4 @@
Follow this link to confirm the update of your email from Follow this link to confirm the update of your email from
{{ .Email }} to {{ .NewEmail }}: {{ .Email }} to {{ .NewEmail }}:
</p> </p>
<p><a href="{{ .SiteURL }}/admin/#email_change_token={{ .Token }}">Change Email</a></p> <p><a href="http://{{ .SiteURL }}/admin/#email_change_token={{ .Token }}">Change Email</a></p>
@@ -4,4 +4,4 @@
You have been invited to create a user on {{ .SiteURL }}. Follow You have been invited to create a user on {{ .SiteURL }}. Follow
this link to accept the invite: this link to accept the invite:
</p> </p>
<p><a href="{{ .SiteURL }}/admin/#invite_token={{ .Token }}">Accept the invite</a></p> <p><a href="http://{{ .SiteURL }}/admin/#invite_token={{ .Token }}">Accept the invite</a></p>
@@ -2,4 +2,4 @@
<p>Follow this link to reset the password for your user:</p> <p>Follow this link to reset the password for your user:</p>
<p><a href="{{ .SiteURL }}/admin/#recovery_token={{ .Token }}">Reset Password</a></p> <p><a href="http://{{ .SiteURL }}/admin/#recovery_token={{ .Token }}">Reset Password</a></p>
Binary file not shown.

After

Width:  |  Height:  |  Size: 36 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 65 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 128 KiB

@@ -0,0 +1,72 @@
<?xml version="1.0" encoding="UTF-8"?>
<svg id="Layer_1" data-name="Layer 1" xmlns="http://www.w3.org/2000/svg" viewBox="0 0 40 40">
<defs>
<style>
.cls-1 {
fill: #7ec465;
}
.cls-1, .cls-2, .cls-3, .cls-4, .cls-5, .cls-6, .cls-7, .cls-8, .cls-9, .cls-10, .cls-11 {
stroke-width: 0px;
}
.cls-2 {
fill: #f2695e;
}
.cls-3 {
fill: #fdd92f;
}
.cls-4 {
fill: #023848;
}
.cls-5 {
fill: #fdda32;
}
.cls-6 {
fill: #7dc363;
}
.cls-7 {
fill: #034a61;
}
.cls-8 {
fill: #8e9da9;
}
.cls-9 {
fill: #e7f4fa;
}
.cls-10 {
fill: #71b3e2;
}
.cls-11 {
fill: #73b4e2;
}
</style>
</defs>
<path class="cls-9" d="m2,27.87c0-7.11,0-14.21,0-21.32,0-.83.49-1.52,1.25-1.83.15-.06.31-.07.47-.1.06-.01.12,0,.18,0,10.73,0,21.45,0,32.18,0,.83,0,1.52.5,1.83,1.25.06.15.07.31.1.47.01.06,0,.12,0,.18v21.35c-.11.1-.24.11-.37.12-.12,0-.24,0-.36,0-11.51,0-23.02,0-34.53,0-.25,0-.51.06-.74-.12Z"/>
<path class="cls-7" d="m2,27.87h36.01c0,.4,0,.8,0,1.2.02.83-.49,1.52-1.25,1.83-.19.07-.4.11-.61.11-3.1,0-6.19,0-9.29,0-.14.12-.31.12-.48.12-4.25,0-8.49,0-12.74,0-.17,0-.34,0-.48-.12-3.1,0-6.19,0-9.29,0-.49,0-.92-.18-1.27-.53-.18-.18-.34-.36-.43-.59-.08-.2-.16-.4-.16-.63,0-.46,0-.92,0-1.38Z"/>
<path class="cls-4" d="m13.15,31.01h13.7c-.01.16.09.29.16.4.27.44.52.91.79,1.35.29.46.53.96.84,1.4.17.24.28.51.42.77.04.07.04.15.08.26-.08.05-.17.1-.31.18H11.07c-.05-.06-.11-.12-.18-.19-.05-.2.09-.36.18-.53.12-.23.26-.45.41-.67.25-.39.48-.8.71-1.2.22-.38.46-.75.66-1.15.09-.17.19-.33.28-.5.02-.04.02-.09.03-.13Z"/>
<path class="cls-10" d="m6.81,17.45c-.4-.22-.71-.56-1-.91-.51-.6-.9-1.28-1.11-2.05-.11-.41-.23-.82-.22-1.26,0-.16-.1-.31-.07-.48.06-.4.07-.81.16-1.21.08-.38.2-.76.36-1.12.36-.81.89-1.5,1.56-2.08.25-.22.52-.43.82-.58.28-.14.53-.35.86-.38.45-.26.95-.3,1.45-.38.21-.03.42-.06.63-.06.07.07.1.16.1.26,0,1.02.05,2.04-.02,3.06-.05.18-.21.18-.35.2-.99.16-1.66.72-1.95,1.67-.27.9-.01,1.69.64,2.36.06.06.13.11.15.2.02.11-.03.19-.09.27-.59.79-1.18,1.58-1.77,2.36-.04.05-.08.09-.14.12Z"/>
<path class="cls-10" d="m35.02,26.91c-.1.07-.22.16-.33.24h-1.93c-.09-.09-.2-.2-.3-.3,0-3.55,0-7.08,0-10.6,0-.11,0-.21.02-.32.03-.17.18-.23.33-.31h1.87c.11.08.22.16.33.24v11.04Z"/>
<path class="cls-3" d="m10.23,10.38v-3.43c.33.02.67.01.99.08.38.08.76.19,1.14.32.51.18.97.45,1.39.77.29.23.63.41.81.77.08.16.25.28.37.43.26.32.45.67.62,1.02.17.37.29.75.39,1.15.15.61.15,1.22.13,1.84,0,.28-.06.58-.15.85-.09.28-.15.57-.3.82,0,.02-.04.02-.06.03-.12.09-.23.02-.34-.03-.81-.35-1.62-.71-2.44-1.05-.11-.05-.22-.1-.27-.22-.05-.16,0-.31.03-.46.3-1.47-.81-2.75-2.07-2.8-.08,0-.17-.02-.23-.09Z"/>
<path class="cls-11" d="m29.23,16.23c.41-.4.8-.81,1.23-1.17.29-.24.53-.52.82-.75.36-.28.67-.62,1-.93.05-.05.12-.09.23-.16-.39.02-.74.02-.97-.32-.03-.1-.06-.2-.1-.32.11-.18.13-.45.55-.52.28,0,.72-.03,1.16.01.29.03.59.06.88.07.09,0,.19.07.27.12.08.05.14.13.2.19,0,.28,0,.53-.06.79-.05.23-.08.47-.15.7-.04.14-.04.3-.06.44-.02.16-.11.31-.09.48-.09.08-.17.17-.24.23-.1.03-.17.05-.28.09-.12-.04-.28-.1-.4-.15-.23-.21-.15-.46-.18-.71-.19.15-.37.28-.54.43-.4.36-.79.74-1.18,1.11-.45.43-.9.86-1.36,1.29-.14.13-.29.23-.46.36-.06.02-.14.04-.25.08-.12-.05-.3-.07-.41-.17-.47-.4-1.01-.69-1.5-1.04-.63-.45-1.29-.87-1.94-1.29-.16-.1-.33-.19-.52-.3-.18.22-.34.42-.5.62-.02.03-.03.06-.05.09-.08.11-.17.22-.26.33-.12.15-.26.28-.35.44-.09.15-.2.29-.3.42-.21.25-.41.51-.6.77-.29.38-.58.75-.86,1.13-.2.26-.41.51-.6.78-.21.28-.46.52-.63.83-.06.11-.19.17-.32.18-.12,0-.24,0-.35,0-.27-.07-.3-.32-.41-.52.03-.1.06-.19.11-.33.31-.38.63-.81.95-1.24.19-.25.39-.49.58-.75.29-.38.58-.75.86-1.13.18-.25.39-.49.58-.75.29-.39.6-.76.88-1.16.23-.33.5-.63.77-.92.24-.26.47-.26.77-.06.39.27.77.53,1.16.8.9.61,1.8,1.22,2.7,1.83.05.04.12.05.23.1Z"/>
<path class="cls-6" d="m12.56,13.73c.06.05.11.12.18.13.33.09.62.26.93.38.48.18.93.44,1.42.6.16.05.28.2.46.2,0,.07,0,.15-.03.21-.13.26-.24.55-.42.77-.28.34-.51.71-.84,1-.52.45-1.07.87-1.72,1.12-.38.14-.75.34-1.17.33-.16.14-.37,0-.51.08-.19.11-.37.04-.56.06-.18.02-.38.05-.54,0-.26-.08-.53-.04-.78-.13-.23-.09-.48-.14-.71-.21-.25-.08-.47-.22-.71-.32-.28-.12-.54-.27-.76-.49,0-.04,0-.1.03-.13.39-.47.74-.97,1.1-1.46.27-.37.56-.73.84-1.1.16-.06.27.06.4.12.82.39,2.07.34,2.8-.51.11-.13.2-.27.3-.41.07-.1.11-.25.29-.22Z"/>
<path class="cls-1" d="m25.79,17.67c.19.15.26.34.26.59,0,2.77,0,5.53,0,8.3,0,.27-.06.48-.38.59h-1.93s-.1-.1-.15-.15c-.12-.12-.18-.26-.18-.44,0-1,0-1.99,0-2.99,0-1.75,0-3.5,0-5.24,0-.13.01-.25.07-.37.08-.17.18-.28.37-.28.4,0,.8,0,1.2,0h.72Z"/>
<path class="cls-5" d="m30.21,27.15h-1.97c-.09-.09-.2-.2-.3-.3,0-2.18,0-4.33,0-6.47,0-.27.04-.49.39-.59h1.93c.1.12.21.26.31.38,0,2.08,0,4.15,0,6.21,0,.32-.05.59-.36.77Z"/>
<path class="cls-8" d="m5,25.79c-.03-.1-.06-.2-.1-.31.06-.12.13-.25.19-.38.14-.15.31-.14.49-.14,3.21,0,6.41,0,9.65,0,.1.12.21.25.3.35.09.44-.08.69-.52.82-3.04,0-6.15,0-9.26,0-.31,0-.56-.05-.75-.34Z"/>
<path class="cls-8" d="m4.9,20.67c.07-.14.13-.26.19-.37.14-.15.31-.15.49-.15,3.18,0,6.36,0,9.54,0,.18.09.36.17.41.4.03.19.04.39-.11.58-.1.05-.22.11-.37.19-3.14,0-6.31,0-9.47,0-.27,0-.43-.16-.57-.31-.04-.13-.07-.23-.11-.34Z"/>
<path class="cls-8" d="m15.12,22.55c.38.15.48.42.43.78-.15.26-.39.39-.7.39-3.15,0-6.29,0-9.46,0-.12-.09-.26-.2-.36-.27-.05-.14-.08-.24-.12-.36.13-.17.11-.47.41-.53h9.8Z"/>
<path class="cls-2" d="m18.98,22.86c.1-.07.22-.16.33-.24h1.93c.09.09.2.2.3.3v3.93c-.09.09-.2.2-.3.3h-1.93c-.11-.08-.22-.16-.33-.24v-4.05Z"/>
<path class="cls-8" d="m27.98,11.84c.1.1.2.21.32.32v.44c-.12.33-.39.41-.7.41-2.83,0-5.66,0-8.49,0-.31,0-.58-.07-.71-.42v-.42c.11-.1.22-.2.35-.32h9.23Z"/>
<path class="cls-8" d="m18.4,7.34c.13-.1.25-.2.38-.3h9.15c.12.09.26.19.39.3,0,.1,0,.19,0,.27,0,.16-.04.29-.15.41-.11.12-.27.12-.4.17-.06.03-.14,0-.22,0-2.79,0-5.59,0-8.38,0-.14,0-.27,0-.41-.07-.21-.09-.34-.21-.36-.42,0-.12,0-.24,0-.37Z"/>
<path class="cls-8" d="m18.4,9.79c.11-.11.24-.24.35-.35h9.17c.12.09.26.2.39.3,0,.16.02.31,0,.45-.03.2-.29.39-.54.41-.07,0-.15,0-.22,0-2.79,0-5.59,0-8.38,0-.12,0-.24,0-.38-.06-.29-.14-.38-.21-.39-.49,0-.07,0-.14,0-.25Z"/>
<path class="cls-9" d="m12.56,13.73c-.13.02-.14.14-.19.23-.28.56-.75.89-1.29,1.13-.15.07-.33.07-.5.09-.12.01-.24,0-.36,0-.48.03-.91-.08-1.3-.36-.04-.03-.1-.05-.15-.07-.02-.03-.03-.08-.05-.08-.27-.1-.39-.36-.52-.58-.13-.22-.25-.44-.31-.7-.09-.35-.03-.68-.06-1.02.16-.4.23-.84.58-1.14.23-.2.41-.43.7-.58.36-.18.72-.29,1.12-.27.41-.05.79.06,1.14.24.56.28.93.75,1.18,1.31.04.1-.04.22.08.3-.03.28.17.54.02.85-.09.19-.06.43-.09.66Z"/>
</svg>

After

Width:  |  Height:  |  Size: 6.5 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 118 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 26 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 104 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 199 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 80 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 240 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.1 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 159 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.3 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 195 KiB

Some files were not shown because too many files have changed in this diff Show More