diff --git a/.github/workflows/internal-dead-links.yml b/.github/workflows/internal-dead-links.yml index 14ba77fb0..0ecf31cf0 100644 --- a/.github/workflows/internal-dead-links.yml +++ b/.github/workflows/internal-dead-links.yml @@ -27,11 +27,11 @@ jobs: export PATH="${CURRENT_DIR}/dart-sass:${PATH}" cd qdrant-landing && hugo --gc -b 'http://localhost:1313' && hugo serve & sleep 5 # wait for server to start - - name: Link Checker + - name: Internal Links Check id: lychee uses: lycheeverse/lychee-action@v1.8.0 with: - args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' qdrant-landing/public/ + args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' --base http://localhost:1313/ qdrant-landing/public/ fail: true env: GITHUB_TOKEN: ${{secrets.GITHUB_TOKEN}} diff --git a/qdrant-landing/config.toml b/qdrant-landing/config.toml index 41afb6c92..efad10fd5 100644 --- a/qdrant-landing/config.toml +++ b/qdrant-landing/config.toml @@ -101,207 +101,6 @@ disableKinds = ["taxonomy", "term"] category = "categories" example = "examples" -[menu] - -[[menu.main]] - identifier = "product" - name = "Product" - weight = 1 - [menu.main.params] - in_header = true - in_footer = true - - [[menu.main]] - identifier = "use-case" - name = "Use cases" - weight = 1 - parent = "product" - url = "/use-cases/" - - [[menu.main]] - identifier = "solutions" - name = "Solutions" - weight = 2 - parent = "product" - url = "/solutions/" - - - [[menu.main]] - identifier = "benchmarks" - name = "Benchmarks" - weight = 3 - parent = "product" - url = "/benchmarks/" - - [[menu.main]] - identifier = "demo" - name = "Demos" - weight = 4 - parent = "product" - url = "/demo/" - - [[menu.main]] - identifier = "pricing" - name = "Pricing" - weight = 5 - parent = "product" - url = "/pricing/" - -[[menu.main]] - identifier = "resources" - name = "Resources" - weight = 2 - [menu.main.params] - in_header = true - in_footer = false - - [[menu.main]] - identifier = "documentation" - name = "Documentation" - parent = "resources" - weight = 1 - url = "/documentation/" - - [[menu.main]] - identifier = "articles" - name = "Articles" - weight = 4 - parent = "resources" - url = "/articles/" - - [[menu.main]] - identifier = "blog" - name = "Blog" - weight = 5 - parent = "resources" - url = "/blog/" - - [[menu.main]] - identifier = "roadmap" - name = "Roadmap" - weight = 6 - parent = "resources" - url = "https://qdrant.to/roadmap" - [menu.main.params] - external = true - - [[menu.main]] - identifier = "changelog" - name = "Changelog" - weight = 7 - parent = "resources" - url = "https://github.com/qdrant/qdrant/releases" - [menu.main.params] - external = true - - [[menu.main]] - identifier = "trust-center" - name = "Trust Center" - weight = 8 - parent = "resources" - url = "http://qdrant.to/trust-center" - [menu.main.params] - external = true - -[[menu.main]] - identifier = "community" - name = "Community" - weight = 4 - [menu.main.params] - in_header = true - in_footer = true - - [[menu.main]] - identifier = "github" - name = "Github" - weight = 1 - parent = "community" - url = "https://github.com/qdrant/qdrant" - pre = "" - [menu.main.params] - external = true - - [[menu.main]] - identifier = "discord" - name = "Discord" - weight = 2 - parent = "community" - url = "https://qdrant.to/discord" - pre = "" - [menu.main.params] - external = true - - [[menu.main]] - identifier = "twitter" - name = "Twitter" - weight = 3 - parent = "community" - url = "https://qdrant.to/twitter" - pre = "" - [menu.main.params] - external = true - - [[menu.main]] - identifier = "newsletter" - name = "Newsletter" - weight = 4 - parent = "community" - url = "/subscribe/" - pre = "" - - [[menu.main]] - identifier = "contact" - name = "Contact us" - weight = 5 - parent = "community" - url = "https://qdrant.to/contact-us" - pre = "" - [menu.main.params] - external = true - -[[menu.main]] - identifier = "company" - name = "Company" - weight = 4 - [menu.main.params] - in_header = false - in_footer = true - - [[menu.main]] - identifier = "jobs" - name = "Jobs" - weight = 1 - parent = "company" - url = "https://qdrant.join.com" - - [[menu.main]] - identifier = "privacy-policy" - name = "Privacy Policy" - weight = 2 - parent = "company" - url = "/legal/privacy-policy/" - - [[menu.main]] - identifier = "terms" - name = "Terms" - weight = 3 - parent = "company" - url = "/legal/terms_and_conditions/" - - [[menu.main]] - identifier = "impressum" - name = "Impressum" - weight = 4 - parent = "company" - url = "/legal/impressum/" - - [[menu.main]] - identifier = "credits" - name = "Credits" - weight = 5 - parent = "company" - url = "/legal/credits/" - [markup] [markup.goldmark.renderer] unsafe=true diff --git a/qdrant-landing/content/articles/dimension-reduction-qsoc.md b/qdrant-landing/content/articles/dimension-reduction-qsoc.md new file mode 100644 index 000000000..09b8ae707 --- /dev/null +++ b/qdrant-landing/content/articles/dimension-reduction-qsoc.md @@ -0,0 +1,126 @@ +--- +title: Qdrant Summer of Code 2024 - WASM based Dimension Reduction +short_description: QSOC'24 WASM based Dimension Reduction +description: My journey as a Qdrant Summer of Code 2024 participant working on enhancing vector visualization using WebAssembly (WASM) based dimension reduction. +preview_dir: /articles_data/dimension-reduction-qsoc/preview +small_preview_image: /articles_data/dimension-reduction-qsoc/icon.svg +social_preview_image: /articles_data/dimension-reduction-qsoc/preview/social_preview.jpg +weight: -10 +author: Jishan Bhattacharya +author_link: https://www.linkedin.com/in/j16n/ +date: 2024-08-31T10:39:48.312Z +draft: false +keywords: + + - dimension reduction + - web assembly + - qsoc'24 + - vector similarity + - tsne + - qdrant data visualization +--- + + + +## Introduction + +Hello, everyone! I'm Jishan Bhattacharya, and I had the incredible opportunity to intern at Qdrant this summer as part of the Qdrant Summer of Code 2024. Under the mentorship of [Andrey Vasnetsov](https://www.linkedin.com/in/andrey-vasnetsov-75268897/), I dived into the world of performance optimization, focusing on enhancing vector visualization using WebAssembly (WASM). In this article, I'll share the insights, challenges, and accomplishments from my journey — one filled with learning, experimentation, and plenty of coding adventures. + + +## Project Overview + +Qdrant is a robust vector database and search engine designed to store vector data and perform tasks like similarity search and clustering. One of its standout features is the ability to visualize high-dimensional vectors in a 2D space. However, the existing implementation faced performance bottlenecks, especially when scaling to large datasets. My mission was to tackle this challenge by leveraging a WASM-based solution for dimensionality reduction in the visualization process. + + +## Learnings & Challenges + +Our weapon of choice was Rust, paired with WASM, and we employed the t-SNE algorithm for dimensionality reduction. For those unfamiliar, t-SNE (t-Distributed Stochastic Neighbor Embedding) is a technique that helps visualize high-dimensional data by projecting it into two or three dimensions. It operates in two main steps: + +1. **Computing Pairwise Similarity:** This step involves calculating the similarity between each pair of data points in the original high-dimensional space. + +2. **Iterative Optimization:** The second step is iterative, where the embedding is refined using gradient descent. Here, the similarity matrix from the first step plays a crucial role. + +At the outset, Andrey tasked me with rewriting the existing JavaScript implementation of t-SNE in Rust, introducing multi-threading along the way. Setting up WASM with Vite for multi-threaded execution was no small feat, but the effort paid off. The resulting Rust implementation outperformed the single-threaded JavaScript version, although it still struggled with large datasets. + +Next came the challenge of optimizing the algorithm further. A key aspect of t-SNE's first step is finding the nearest neighbors for each data point, which requires an efficient data structure. I opted for a [Vantage Point Tree](https://en.wikipedia.org/wiki/Vantage-point_tree) (also known as a Ball Tree) to speed up this process. As for the second step, while it is inherently sequential, there was still room for improvement. I incorporated Barnes-Hut approximation to accelerate the gradient calculation. This method approximates the forces between points in low dimensional space, making the process more efficient. + +To illustrate, imagine dividing a 2D space into quadrants, each containing multiple points. Every quadrant is again subdivided into four quadrants. This is done until every point belongs to a single cell. + +{{< figure + src="/articles_data/dimension-reduction-qsoc/barnes_hut.png" + caption="Barnes-Hut Approximation" + alt="Calculating the resultant force on red point using Barnes-Hut approximation" +>}} + +We then calculate the center of mass for each cell represented by a blue circle as shown in the figure. Now let’s say we want to find all the forces, represented by dotted lines, on the red point. Barnes Hut’s approximation states that for points that are sufficiently distant, instead of computing the force for each individual point, we use the center of mass as a proxy, significantly reducing the computational load. This is represented by the blue dotted line in the figure. + +These optimizations made a remarkable difference — Barnes-Hut t-SNE was eight times faster than the exact t-SNE for 10,000 vectors. + +{{< figure + src="/articles_data/dimension-reduction-qsoc/rust_rewrite.jpg" + caption="Exact t-SNE - Total time: 884.728s" + alt="Image of visualizing 10,000 vectors using exact t-SNE which took 884.728s" +>}} + +{{< figure + src="/articles_data/dimension-reduction-qsoc/rust_bhtsne.jpg" + caption="Barnes-Hut t-SNE - Total time: 104.191s" + alt="Image of visualizing 10,000 vectors using Barnes-Hut t-SNE which took 110.728s" +>}} + +Despite these improvements, the first step of the algorithm was still a bottleneck, leading to noticeable delays and blank screens. I experimented with approximate nearest neighbor algorithms, but the performance gains were minimal. After consulting with my mentor, we decided to compute the nearest neighbors on the server side, passing the distance matrix directly to the visualization process instead of the raw vectors. + +While waiting for the distance-matrix API to be ready, I explored further optimizations. I observed that the worker thread sent results to the main thread for rendering at specific intervals, causing unnecessary delays due to serialization and deserialization. + +{{< figure + src="/articles_data/dimension-reduction-qsoc/channels.png" + caption="Serialization and Deserialization Overhead" + alt="Image showing serialization and deserialization overhead due to message passing between threads" +>}} + +To address this, I implemented a `SharedArrayBuffer`, allowing the main thread to access changes made by the worker thread instantly. This change led to noticeable improvements. + +Additionally, the previous architecture resulted in choppy animations due to the fixed intervals at which the worker thread sent results. + +{{< figure + src="/articles_data/dimension-reduction-qsoc/prev_arch.png" + caption="Previous architecture with fixed intervals" + alt="Image showing the previous architecture of the frontend with fixed intervals for sending results" +>}} + +I introduced a "rendering-on-demand" approach, where the main thread would signal the worker thread when it was ready to render the next result. This created smoother, more responsive animations. + +{{< figure + src="/articles_data/dimension-reduction-qsoc/curr_arch.png" + caption="Current architecture with rendering-on-demand" + alt="Image showing the current architecture of the frontend with rendering-on-demand approach" +>}} + +With these optimizations in place, the final step was wrapping up the project by creating a Node.js [package](https://www.npmjs.com/package/wasm-dist-bhtsne). This package exposed the necessary interfaces to accept the distance matrix, perform calculations, and return the results, making the solution easy to integrate into various projects. + + +## Areas for Improvement + +While reflecting on this transformative journey, there are still areas that offer room for improvement and future enhancements: + +1. **Payload Parsing:** When requesting a large number of vectors, parsing the payload on the main thread can make the user interface unresponsive. Implementing a faster parser could mitigate this issue. + +2. **Direct Data Requests:** Allowing the worker thread to request data directly could eliminate the initial transfer of data from the main thread, speeding up the overall process. + +3. **Chart Library Optimization:** Profiling revealed that nearly 80% of the time was spent on the Chart.js update function. Switching to a WebGL-accelerated chart library could dramatically improve performance, especially for large datasets. +{{< figure + src="/articles_data/dimension-reduction-qsoc/profiling.png" + caption="Profiling Result" + alt="Image showing profiling results with 80% time spent on Chart.js update function" +>}} + + +## Conclusion + +Participating in the Qdrant Summer of Code 2024 was a deeply rewarding experience. I had the chance to push the boundaries of my coding skills while exploring new technologies like Rust and WebAssembly. I'm incredibly grateful for the guidance and support from my mentor and the entire Qdrant team, who made this journey both educational and enjoyable. + +This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide. + +This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide. + +Thank you for joining me on this coding adventure. I hope you found something valuable in my journey, and I look forward to sharing more exciting projects with you in the future. Happy coding! \ No newline at end of file diff --git a/qdrant-landing/content/articles/qdrant-1.7.x.md b/qdrant-landing/content/articles/qdrant-1.7.x.md index 3fe9c617d..3a1efaa93 100644 --- a/qdrant-landing/content/articles/qdrant-1.7.x.md +++ b/qdrant-landing/content/articles/qdrant-1.7.x.md @@ -51,7 +51,7 @@ Things have changed since then, as so many of you wanted a single tool for spars If you're coming across the topic of sparse vectors for the first time, our [Brief History of Search](/documentation/overview/vector-search/) explains the difference between sparse and dense vectors. -Check out the [sparse vectors article](../sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users. +Check out the [sparse vectors article](/articles/sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users. ### Discovery API diff --git a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md index a8f6447fe..7e2f7c401 100755 --- a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md +++ b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md @@ -21,7 +21,7 @@ keywords: In today's fast-paced, information-rich world, AI is revolutionizing knowledge management. The systematic process of capturing, distributing, and effectively using knowledge within an organization is one of the fields in which AI provides exceptional value today. -> The potential for AI-powered knowledge management increases when leveraging Retrieval Augmented Generation (RAG), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases. +> The potential for AI-powered knowledge management increases when leveraging [Retrieval Augmented Generation (RAG)](https://qdrant.tech/rag/rag-evaluation-guide/), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases. This process enhances the accuracy, relevance, and reliability of generated text, thereby mitigating the risk of faulty, incorrect, or nonsensical results sometimes associated with traditional LLMs. This method not only ensures that the answers are contextually relevant but also up-to-date, reflecting the latest insights and data available. @@ -35,7 +35,7 @@ In this article, we’ll break down a RAG Optimization workflow experiment that Alongside Qdrant we will use Quotient, which provides a seamless way to evaluate your RAG implementation, accelerating and improving the experimentation process. -[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build evaluation frameworks and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users. +[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build [evaluation frameworks](https://qdrant.tech/rag/rag-evaluation-guide/) and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users. > 💡 The [accompanying notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient) for this exercise can be found on GitHub for future reference. @@ -50,11 +50,11 @@ Let us walk you through how we arrived at these findings! ## Building a RAG pipeline -To evaluate a RAG pipeline , we will have to build a RAG Pipeline first. In the interest of simplicity, we are building a Naive RAG in this article. There are certainly other versions of RAG : +To evaluate a RAG pipeline, we will have to build a RAG Pipeline first. In the interest of simplicity, we are building a Naive RAG in this article. There are certainly other versions of RAG : ![shades_of_rag.png](/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/shades_of_rag.png) -The illustration below depicts how we can leverage a RAG Evaluation framework to assess the quality of RAG Application. +The illustration below depicts how we can leverage a [RAG Evaluation framework](https://qdrant.tech/rag/rag-evaluation-guide/) to assess the quality of RAG Application. ![qdrant_and_quotient.png](/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/qdrant_and_quotient.png) @@ -686,4 +686,4 @@ This iterative process demonstrates how, starting from scratch, continual evalua > A workshop version of this article is [available on YouTube](https://www.youtube.com/watch?v=3MEMPZR1aZA). Follow along using our [GitHub notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient). - \ No newline at end of file + diff --git a/qdrant-landing/content/articles/semantic-cache-ai-data-retrieval.md b/qdrant-landing/content/articles/semantic-cache-ai-data-retrieval.md index 1ea08a35b..191668fca 100644 --- a/qdrant-landing/content/articles/semantic-cache-ai-data-retrieval.md +++ b/qdrant-landing/content/articles/semantic-cache-ai-data-retrieval.md @@ -24,7 +24,7 @@ tags: **Semantic cache** is a method of retrieval optimization, where similar queries instantly retrieve the same appropriate response from a knowledge base. -Semantic cache differs from traditional caching methods. In computing, **cache** refers to high-speed memory that efficiently stores frequently accessed data. In the context of vector databases, a **semantic cache** improves AI application performance by storing previously retrieved results along with the conditions under which they were computed. This allows the application to reuse those results when the same or similar conditions occur again, rather than finding them from scratch. +Semantic cache differs from traditional caching methods. In computing, **cache** refers to high-speed memory that efficiently stores frequently accessed data. In the context of [vector databases](/articles/what-is-a-vector-database/), a **semantic cache** improves AI application performance by storing previously retrieved results along with the conditions under which they were computed. This allows the application to reuse those results when the same or similar conditions occur again, rather than finding them from scratch. > The term **"semantic"** implies that the cache takes into account the meaning or semantics of the data or computation being cached, rather than just its syntactic representation. This can lead to more efficient caching strategies that exploit the structure or relationships within the data or computation. @@ -40,9 +40,9 @@ In this blog and video, we will walk you through how to use Qdrant to implement Semantic cache is increasingly used in Retrieval-Augmented Generation (RAG) applications. In RAG, when a user asks a question, we embed it and search our vector database, either by using keyword, semantic, or hybrid search methods. The matched context is then passed to a Language Model (LLM) along with the prompt and user question for response generation. -Qdrant is recommended for setting up semantic cache as semantically evaluates the response. When semantic cache is implemented, we store common questions and their corresponding answers in a key-value cache. This way, when a user asks a question, we can retrieve the response from the cache if it already exists. +Qdrant is recommended for setting up semantic cache as semantically [evaluates](https://qdrant.tech/rag/rag-evaluation-guide/) the response. When semantic cache is implemented, we store common questions and their corresponding answers in a key-value cache. This way, when a user asks a question, we can retrieve the response from the cache if it already exists. -**Diagram:** Semantic cache improves RAG by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers. +**Diagram:** Semantic cache improves [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers. ![Alt Text](/articles_data/semantic-cache-ai-data-retrieval/semantic-cache.gif) diff --git a/qdrant-landing/content/articles/what-is-quantization.md b/qdrant-landing/content/articles/what-is-quantization.md new file mode 100644 index 000000000..bc570c297 --- /dev/null +++ b/qdrant-landing/content/articles/what-is-quantization.md @@ -0,0 +1,531 @@ +--- +title: "What is Vector Quantization?" +draft: false +slug: what-is-vector-quantization +short_description: What is Vector Quantization? Methods & Examples +description: In this article, we'll teach you about compression methods like Scalar, Product, and Binary Quantization. Learn how to choose the best method for your specific application. +preview_dir: /articles_data/what-is-vector-quantization/preview +weight: -210 +social_preview_image: /articles_data/what-is-vector-quantization/preview/social-preview.jpg +date: 2024-09-25T09:29:33-03:00 +author: Sabrina Aquino +featured: true +tags: + - vector-search + - vector-quantization + - binary quantization + - product quantization + - scalar quantization + - vector compression + +--- + +Vector quantization is a data compression technique used to reduce the size of high-dimensional data. Compressing vectors reduces memory usage while maintaining nearly all of the essential information. This method allows for more efficient storage and faster search operations, particularly in large datasets. + +When working with high-dimensional vectors, such as embeddings from providers like OpenAI, a single 1536-dimensional vector requires **6 KB of memory**. + +1536-dimensional vector size is 6 KB + +With 1 million vectors needing around 6 GB of memory, as your dataset grows to multiple **millions of vectors**, the memory and processing demands increase significantly. + +To understand why this process is so computationally demanding, let's take a look at the nature of the [HNSW index](https://qdrant.tech/documentation/concepts/indexing/#vector-index). + +The **HNSW (Hierarchical Navigable Small World) index** organizes vectors in a layered graph, connecting each vector to its nearest neighbors. At each layer, the algorithm narrows down the search area until it reaches the lower layers, where it efficiently finds the closest matches to the query. + +HNSW Search visualization + +Each time a new vector is added, the system must determine its position in the existing graph, a process similar to searching. This makes both inserting and searching for vectors complex operations. + +One of the key challenges with the HNSW index is that it requires a lot of **random reads** and **sequential traversals** through the graph. This makes the process computationally expensive, especially when you're dealing with millions of high-dimensional vectors. + +The system has to jump between various points in the graph in an unpredictable manner. This unpredictability makes optimization difficult, and as the dataset grows, the memory and processing requirements increase significantly. + +HNSW Search visualization + +Since vectors need to be stored in **fast storage** like **RAM** or **SSD** for low-latency searches, as the size of the data grows, so does the cost of storing and processing it efficiently. + +**Quantization** offers a solution by compressing vectors to smaller memory sizes, making the process more efficient. + +There are several methods to achieve this, and here we will focus on three main ones: + +Types of Quantization: 1. Scalar Quantization, 2. Product Quantization, 3. Binary Quantization + +## 1. What is Scalar Quantization? + +![](/articles_data/what-is-vector-quantization/astronaut-mars.jpg) + +In Qdrant, each dimension is represented by a `float32` value, which uses **4 bytes** of memory. When using [Scalar Quantization](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization), we map our vectors to a range that the smaller `int8` type can represent. An `int8` is only **1 byte** and can represent 256 values (from -128 to 127, or 0 to 255). This results in a **75% reduction** in memory size. + +For example, if our data lies in the range of -1.0 to 1.0, Scalar Quantization will transform these values to a range that `int8` can represent, i.e., within -128 to 127. The system **maps** the `float32` values into this range. + +Here's a simple linear example of what this process looks like: + +![Scalar Quantization example](/articles_data/what-is-vector-quantization/scalar-quant.png) + +To set up Scalar Quantization in Qdrant, you need to include the `quantization_config` section when creating or updating a collection: + +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 128, + "distance": "Cosine" + }, + "quantization_config": { + "scalar": { + "type": "int8", + "quantile": 0.99, + "always_ram": true + } + } +} +``` + +```python +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=128, distance=models.Distance.COSINE), + quantization_config=models.ScalarQuantization( + scalar=models.ScalarQuantizationConfig( + type=models.ScalarType.INT8, + quantile=0.99, + always_ram=True, + ), + ), +) +``` + +The `quantile` parameter is used to calculate the quantization bounds. For example, if you specify a `0.99` quantile, the most extreme 1% of values will be excluded from the quantization bounds. + +This parameter only affects the resulting precision, not the memory footprint. You can adjust it if you experience a significant decrease in search quality. + +Scalar Quantization is a great choice if you're looking to boost search speed and compression without losing much accuracy. It also slightly improves performance, as distance calculations (such as dot product or cosine similarity) using `int8` values are computationally simpler than using `float32` values. + +While the performance gains of Scalar Quantization may not match those achieved with Binary Quantization (which we'll discuss later), it remains an excellent default choice when Binary Quantization isn’t suitable for your use case. + +## 2. What is Binary Quantization? + +![Astronaut in surreal white environment](/articles_data/what-is-vector-quantization/astronaut-white-surreal.jpg) + +[Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) is an excellent option if you're looking to **reduce memory** usage while also achieving a significant **boost in speed**. It works by converting high-dimensional vectors into simple binary (0 or 1) representations. + +- Values greater than zero are converted to 1. +- Values less than or equal to zero are converted to 0. + +Let's consider our initial example of a 1536-dimensional vector that requires **6 KB** of memory (4 bytes for each `float32` value). + +After Binary Quantization, each dimension is reduced to 1 bit (1/8 byte), so the memory required is: + +$$ +\frac{1536 \text{ dimensions}}{8 \text{ bits per byte}} = 192 \text{ bytes} +$$ + +This leads to a **32x** memory reduction. + +Binary Quantization example + +Qdrant automates the Binary Quantization process during indexing. As vectors are added to your collection, each 32-bit floating-point component is converted into a binary value according to the configuration you define. + +Here’s how you can set it up: + +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 1536, + "distance": "Cosine" + }, + "quantization_config": { + "binary": { + "always_ram": true + } + } +} +``` + +```python +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), + quantization_config=models.BinaryQuantization( + binary=models.BinaryQuantizationConfig( + always_ram=True, + ), + ), +) +``` + +Binary Quantization is by far the quantization method that provides the most significant processing **speed gains** compared to Scalar and Product Quantizations. This is because the binary representation allows the system to use highly optimized CPU instructions, such as [XOR](https://en.wikipedia.org/wiki/XOR_gate#:~:text=XOR%20represents%20the%20inequality%20function,the%20other%20but%20not%20both%22) and [Popcount](https://en.wikipedia.org/wiki/Hamming_weight), for fast distance computations. + +It can speed up search operations by **up to 40x**, depending on the dataset and hardware. + +Not all models are equally compatible with Binary Quantization, and in the comparison above, we are only using models that are compatible. Some models may experience a greater loss in accuracy when quantized. We recommend using Binary Quantization with models that have **at least 1024 dimensions** to minimize accuracy loss. + +The models that have shown the best compatibility with this method include: + +- **OpenAI text-embedding-ada-002** (1536 dimensions) +- **Cohere AI embed-english-v2.0** (4096 dimensions) + +These models demonstrate minimal accuracy loss while still benefiting from substantial speed and memory gains. + +Even though Binary Quantization is incredibly fast and memory-efficient, the trade-offs are in **precision** and **model compatibility**, so you may need to ensure search quality using techniques like oversampling and rescoring. + +If you're interested in exploring Binary Quantization in more detail—including implementation examples, benchmark results, and usage recommendations—check out our dedicated article on [Binary Quantization - Vector Search, 40x Faster](https://qdrant.tech/articles/binary-quantization/). + +## 3. What is Product Quantization? + +![](/articles_data/what-is-vector-quantization/astronaut-centroids.jpg) + +[Product Quantization](https://qdrant.tech/documentation/guides/quantization/#product-quantization) is a method used to compress high-dimensional vectors by representing them with a smaller set of representative points. + +The process begins by splitting the original high-dimensional vectors into smaller **sub-vectors.** Each sub-vector represents a segment of the original vector, capturing different characteristics of the data. + +Creation of the Sub-vector + +For each sub-vector, a separate **codebook** is created, representing regions in the data space where common patterns occur. + +The codebook in Qdrant is trained automatically during the indexing process. As vectors are added to the collection, Qdrant uses your specified quantization settings in the `quantization_config` to build the codebook and quantize the vectors. Here’s how you can set it up: + +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 1024, + "distance": "Cosine" + }, + "quantization_config": { + "product": { + "compression": "x32", + "always_ram": true + } + } +} +``` + + +```python +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=1024, distance=models.Distance.COSINE), + quantization_config=models.ProductQuantization( + product=models.ProductQuantizationConfig( + compression=models.CompressionRatio.X32, + always_ram=True, + ), + ), +) +``` + +Each region in the codebook is defined by a **centroid**, which serves as a representative point summarizing the characteristics of that region. Instead of treating every single data point as equally important, we can group similar sub-vectors together and represent them with a single centroid that captures the general characteristics of that group. + +The centroids used in Product Quantization are determined using the **[K-means clustering algorithm](https://en.wikipedia.org/wiki/K-means_clustering)**. + +Codebook and Centroids example + +Qdrant always selects **K = 256** as the number of centroids in its implementation, based on the fact that 256 is the maximum number of unique values that can be represented by a single byte. + +This makes the compression process efficient because each centroid index can be stored in a single byte. + +The original high-dimensional vectors are quantized by mapping each sub-vector to the nearest centroid in its respective codebook. + +Vectors being mapped to their corresponding centroids example + +The compressed vector stores the index of the closest centroid for each sub-vector. + +Here’s how a 1024-dimensional vector, originally taking up 4096 bytes, is reduced to just 128 bytes by representing it as 128 indexes, each pointing to the centroid of a sub-vector: + +Product Quantization example + +After setting up quantization and adding your vectors, you can perform searches as usual. Qdrant will automatically use the quantized vectors, optimizing both speed and memory usage. Optionally, you can enable rescoring for better accuracy. + + +```http +POST /collections/{collection_name}/points/search +{ + "query": [0.22, -0.01, -0.98, 0.37], + "params": { + "quantization": { + "rescore": true + } + }, + "limit": 10 +} +``` + +```python +client.query_points( + collection_name="my_collection", + query_vector=[0.22, -0.01, -0.98, 0.37], # Your query vector + search_params=models.SearchParams( + quantization=models.QuantizationSearchParams( + rescore=True # Enables rescoring with original vectors + ) + ), + limit=10 # Return the top 10 results +) +``` +Product Quantization can significantly reduce memory usage, potentially offering up to **64x** compression in certain configurations. However, it's important to note that this level of compression can lead to a noticeable drop in quality. + +If your application requires high precision or real-time performance, Product Quantization may not be the best choice. However, if **memory savings** are critical and some accuracy loss is acceptable, it could still be an ideal solution. + +Here’s a comparison of speed, accuracy, and compression for all three methods, adapted from [Qdrant's documentation](https://qdrant.tech/documentation/guides/quantization/#how-to-choose-the-right-quantization-method): + +| Quantization method | Accuracy | Speed | Compression | +|---------------------|----------|------------|-------------| +| Scalar | 0.99 | up to x2 | 4 | +| Product | 0.7 | 0.5 | up to 64 | +| Binary | 0.95* | up to x40 | 32 | + +\* - for compatible models + +For a more in-depth understanding of the benchmarks you can expect, check out our dedicated article on [Product Quantization in Vector Search](https://qdrant.tech/articles/product-quantization/). + +## Rescoring, Oversampling, and Reranking + +When we use quantization methods like Scalar, Binary, or Product Quantization, we're compressing our vectors to save memory and improve performance. However, this compression removes some detail from the original vectors. + +This can slightly reduce the accuracy of our similarity searches because the quantized vectors are approximations of the original data. To mitigate this loss of accuracy, you can use **oversampling** and **rescoring**, which help improve the accuracy of the final search results. + +The original vectors are never deleted during this process, and you can easily switch between quantization methods or parameters by updating the collection configuration at any time. + +Here’s how the process works, step by step: + +### 1. Initial Quantized Search + +When you perform a search, Qdrant retrieves the top candidates using the quantized vectors based on their similarity to the query vector, as determined by the quantized data. This step is fast because we're using the quantized vectors. + +ANN Search with Quantization + +### 2. Oversampling + +Oversampling is a technique that helps compensate for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results. + +You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results (`limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2). + +ANN Search with Quantization and Oversampling + +You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of obtaining high-quality top-K results, especially after rescoring with the original vectors. + +### 3. Rescoring with Original Vectors + +After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query. + +The rescoring process **maps** the quantized vectors to their corresponding original vectors, allowing you to consider factors like context, metadata, or additional relevance that wasn’t included in the initial search, leading to more accurate results. + +![Rescoring with Original Vectors](/articles_data/what-is-vector-quantization/rescoring.png) + +During rescoring, one of the lower-ranked candidates from oversampling might turn out to be a better match than some of the original top-K candidates. + +Even though rescoring uses the original, larger vectors, the process remains much faster because only a very small number of vectors are read. The initial quantized search already identifies the specific vectors to read, rescore, and rerank. + +### 4. Reranking + +With the new similarity scores from rescoring, **reranking** is where the final top-K candidates are determined based on the updated similarity scores. + +For example, in our case with a limit of 4, a candidate that ranked 6th in the initial quantized search might improve its score after rescoring because the original vectors capture more context or metadata. As a result, this candidate could move into the final top 4 after reranking, replacing a less relevant option from the initial search. + +Reranking with Original Vectors + +Here's how you can set it up: + +```http +POST /collections/{collection_name}/points/search + + +{ + "query": [0.22, -0.01, -0.98, 0.37], + "params": { + "quantization": { + "rescore": true, + "oversampling": 2 + } + }, + "limit": 4 +} +``` + +```python +client.query_points( + collection_name="my_collection", + query_vector=[0.22, -0.01, -0.98, 0.37], + search_params=models.SearchParams( + quantization=models.QuantizationSearchParams( + rescore=True, # Enables rescoring with original vectors + oversampling=2 # Retrieves extra candidates for rescoring + ) + ), + limit=4 # Desired number of final results +) +``` + +You can adjust the `oversampling` factor to find the right balance between search speed and result accuracy. + +If quantization is impacting performance in an application that requires high accuracy, combining oversampling with rescoring is a great choice. However, if you need faster searches and can tolerate some loss in accuracy, you might choose to use oversampling without rescoring, or adjust the oversampling factor to a lower value. + +## Distributing Resources Between Disk & Memory + +Qdrant stores both the quantized and original vectors. When you enable quantization, both the original and quantized vectors are stored in RAM by default. You can move the original vectors to disk to significantly reduce RAM usage and lower system costs. Simply enabling quantization is not enough—you need to explicitly move the original vectors to disk by setting `on_disk=True`. + +Here’s an example configuration: + +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 1536, + "distance": "Cosine", + "on_disk": true # Move original vectors to disk + }, + "quantization_config": { + "binary": { + "always_ram": true # Store only quantized vectors in RAM + } + } +} +``` + +```python +client.update_collection( + collection_name="my_collection", + vectors_config=models.VectorParams( + size=1536, + distance=models.Distance.COSINE, + on_disk=True # Move original vectors to disk + ), + quantization_config=models.BinaryQuantization( + binary=models.BinaryQuantizationConfig( + always_ram=True # Store only quantized vectors in RAM + ) + ) +) +``` + +Without explicitly setting `on_disk=True`, you won't see any RAM savings, even with quantization enabled. So, make sure to configure both storage and quantization options based on your memory and performance needs. If your storage has high disk latency, consider disabling rescoring to maintain speed. + +### Speeding Up Rescoring with io_uring + +When dealing with large collections of quantized vectors, frequent disk reads are required to retrieve both original and compressed data for rescoring operations. While `mmap` helps with efficient I/O by reducing user-to-kernel transitions, rescoring can still be slowed down when working with large datasets on disk due to the need for frequent disk reads. + +On Linux-based systems, `io_uring` allows multiple disk operations to be processed in parallel, significantly reducing I/O overhead. This optimization is particularly effective during rescoring, where multiple vectors need to be re-evaluated after the initial search. With io_uring, Qdrant can retrieve and rescore vectors from disk in the most efficient way, improving overall search performance. + +When you perform vector quantization and store data on disk, Qdrant often needs to access multiple vectors in parallel. Without io_uring, this process can be slowed down due to the system’s limitations in handling many disk accesses. + +To enable `io_uring` in Qdrant, add the following to your storage configuration: + +```yaml +storage: + async_scorer: true # Enable io_uring for async storage +``` + +Without this configuration, Qdrant will default to using `mmap` for disk I/O operations. + +For more information and benchmarks comparing io_uring with traditional I/O approaches like mmap, check out [Qdrant's io_uring implementation article.](https://qdrant.tech/articles/io_uring/) + +## Performance of Quantized vs. Non-Quantized Data + +Qdrant uses the quantized vectors by default if they are available. If you want to evaluate how quantization affects your search results, you can temporarily disable it to compare results from quantized and non-quantized searches. To do this, set `ignore: true` in the query: + +```http +POST /collections/{collection_name}/points/query +{ + "query": [0.22, -0.01, -0.98, 0.37], + "params": { + "quantization": { + "ignore": true, + } + }, + "limit": 4 +} +``` + +```python +client.query_points( + collection_name="{collection_name}", + query=[0.22, -0.01, -0.98, 0.37], + search_params=models.SearchParams( + quantization=models.QuantizationSearchParams( + ignore=True + ) + ), +) +``` +### Switching Between Quantization Methods + +Not sure if you’ve chosen the right quantization method? In Qdrant, you have the flexibility to remove quantization and rely solely on the original vectors, adjust the quantization type, or change compression parameters at any time without affecting your original vectors. + +To switch to binary quantization and adjust the compression rate, for example, you can update the collection’s quantization configuration using the `update_collection` method: + +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 1536, + "distance": "Cosine" + }, + "quantization_config": { + "binary": { + "always_ram": true, + "compression_rate": 0.8 # Set the new compression rate + } + } +} +``` + + +```python +client.update_collection( + collection_name="my_collection", + quantization_config=models.BinaryQuantization( + binary=models.BinaryQuantizationConfig( + always_ram=True, # Store only quantized vectors in RAM + compression_rate=0.8 # Set the new compression rate + ) + ), +) +``` + +If you decide to **turn off quantization** and use only the original vectors, you can remove the quantization settings entirely with `quantization_config=None`: + +```http +PUT /collections/my_collection +{ + "vectors": { + "size": 1536, + "distance": "Cosine" + }, + "quantization_config": null # Remove quantization and use original vectors only +} +``` + +```python +client.update_collection( + collection_name="my_collection", + quantization_config=None # Remove quantization and rely on original vectors only +) +``` +## Wrapping Up + +![](/articles_data/what-is-vector-quantization/astronaut-running.jpg) + +Quantization methods like Scalar, Product, and Binary Quantization offer powerful ways to optimize memory usage and improve search performance when dealing with large datasets of high-dimensional vectors. Each method comes with its own trade-offs between memory savings, computational speed, and accuracy. + +Here are some final thoughts to help you choose the right quantization method for your needs: + +| **Quantization Method** | **Key Features** | **When to Use** | +|--------------------------|-------------------------------------------------------------|--------------------------------------------------------------------------------------------| +| **Binary Quantization** | • **Fastest method and most memory-efficient**
• Up to **40x** faster search and **32x** reduced memory footprint | • Use with tested models like OpenAI's `text-embedding-ada-002` and Cohere's `embed-english-v2.0`
• When speed and memory efficiency are critical | +| **Scalar Quantization** | • **Minimal loss of accuracy**
• Up to **4x** reduced memory footprint | • Safe default choice for most applications.
• Offers a good balance between accuracy, speed, and compression. | +| **Product Quantization** | • **Highest compression ratio**
• Up to **64x** reduced memory footprint | • When minimizing memory usage is the top priority
• Acceptable if some loss of accuracy and slower indexing is tolerable | + +### Learn More + +If you want to learn more about improving accuracy, memory efficiency, and speed when using quantization in Qdrant, we have a dedicated [Quantization tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section in our docs that explains all the quantization tips you can use to enhance your results. + +Learn more about optimizing real-time precision with oversampling in Binary Quantization by watching this interview with Qdrant’s CTO, Andrey Vasnetsov: + +
+ +
+ +Stay up-to-date on the latest in [vector search](/advanced-search/) and quantization, share your projects, ask questions, [join our vector search community](https://discord.com/invite/qdrant)! diff --git a/qdrant-landing/content/articles/what-is-rag-in-ai.md b/qdrant-landing/content/articles/what-is-rag-in-ai.md index 0a0730bc6..7cf467b0c 100644 --- a/qdrant-landing/content/articles/what-is-rag-in-ai.md +++ b/qdrant-landing/content/articles/what-is-rag-in-ai.md @@ -34,7 +34,7 @@ While you could be more creative with your prompts, it is only a short-term solu The image above shows how a basic RAG system works. Before forwarding the question to the LLM, we have a layer that searches our knowledge base for the "relevant knowledge" to answer the user query. Specifically, in this case, the spending data from the last month. Our LLM can now generate a **relevant non-hallucinated** response about our budget. -As your data grows, you’ll need efficient ways to identify the most relevant information for your LLM's limited memory. This is where you’ll want a proper way to store and retrieve the specific data you’ll need for your query, without needing the LLM to remember it. +As your data grows, you’ll need [efficient ways](https://qdrant.tech/rag/rag-evaluation-guide/) to identify the most relevant information for your LLM's limited memory. This is where you’ll want a proper way to store and retrieve the specific data you’ll need for your query, without needing the LLM to remember it. **Vector databases** store information as **vector embeddings**. This format supports efficient similarity searches to retrieve relevant data for your query. For example, Qdrant is specifically designed to perform fast, even in scenarios dealing with billions of vectors. @@ -167,7 +167,7 @@ Are you ready to create your own RAG chatbot from the ground up? We have a video * Applying vector similarity search algorithms * Enhancing the efficiency and response quality -After building your RAG chatbot, you'll be able to evaluate its performance against that of a chatbot powered solely by a Large Language Model (LLM). +After building your RAG chatbot, you'll be able to [evaluate its performance](https://qdrant.tech/rag/rag-evaluation-guide/) against that of a chatbot powered solely by a Large Language Model (LLM).
@@ -177,4 +177,4 @@ After building your RAG chatbot, you'll be able to evaluate its performance agai Have a RAG project you want to bring to life? Join our [Discord community](https://discord.gg/qdrant) where we’re always sharing tips and answering questions on vector search and retrieval. -Learn more about how to properly evaluate your RAG responses: [Evaluating Retrieval Augmented Generation - a framework for assessment](https://superlinked.com/vectorhub/evaluating-retrieval-augmented-generation-a-framework-for-assessment). \ No newline at end of file +Learn more about how to properly evaluate your RAG responses: [Evaluating Retrieval Augmented Generation - a framework for assessment](https://superlinked.com/vectorhub/evaluating-retrieval-augmented-generation-a-framework-for-assessment). diff --git a/qdrant-landing/content/blog/azure-marketplace.md b/qdrant-landing/content/blog/azure-marketplace.md index e893b8ef1..da6c9f50d 100644 --- a/qdrant-landing/content/blog/azure-marketplace.md +++ b/qdrant-landing/content/blog/azure-marketplace.md @@ -50,7 +50,7 @@ We're incredibly excited about this collaboration with Azure Marketplace and the Ready to elevate your business with Qdrant? **Click the banner and get started today!** -[![Get Started on Azure Marketplace](cta.png)](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db) +[![Get Started on Azure Marketplace](/blog/azure-marketplace/cta.png)](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db) ### About Qdrant: diff --git a/qdrant-landing/content/blog/case-study-dust.md b/qdrant-landing/content/blog/case-study-dust.md index 87f7b3614..62be7af77 100644 --- a/qdrant-landing/content/blog/case-study-dust.md +++ b/qdrant-landing/content/blog/case-study-dust.md @@ -52,7 +52,8 @@ data usually sits in various SaaS applications across the organization. Dust provides companies with the core platform to execute on their GenAI bet for their teams by deploying LLMs across the organization and providing context -aware AI assistants through RAG. Users can manage so-called data sources within +aware AI assistants through [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) +. Users can manage so-called data sources within Dust and upload files or directly connect to it via APIs to ingest data from tools like Notion, Google Drive, or Slack. Dust then handles the chunking strategy with the embeddings models and performs retrieval augmented generation. diff --git a/qdrant-landing/content/blog/case-study-kern.md b/qdrant-landing/content/blog/case-study-kern.md index d35e60004..466a480e3 100644 --- a/qdrant-landing/content/blog/case-study-kern.md +++ b/qdrant-landing/content/blog/case-study-kern.md @@ -1,7 +1,7 @@ --- draft: false title: "Kern AI & Qdrant: Precision AI Solutions for Finance and Insurance" -short_description: "Transforming customer service in finance and insurance with vector search-based retrieval.

" +short_description: "Transforming customer service in finance and insurance with vector search-based retrieval." description: "Revolutionizing customer service in finance and insurance by leveraging vector search for faster responses and improved operational efficiency." preview_image: /blog/case-study-kern/preview.png social_preview_image: /blog/case-study-kern/preview.png diff --git a/qdrant-landing/content/blog/case-study-nyris.md b/qdrant-landing/content/blog/case-study-nyris.md index 38ef2b95e..4d037de55 100644 --- a/qdrant-landing/content/blog/case-study-nyris.md +++ b/qdrant-landing/content/blog/case-study-nyris.md @@ -1,7 +1,7 @@ --- draft: false title: "Nyris & Qdrant: How Vectors are the Future of Visual Search" -short_description: "Transforming customer service in finance and insurance with vector search-based retrieval.

" +short_description: "Transforming customer service in finance and insurance with vector search-based retrieval." description: "Revolutionizing customer service in finance and insurance by leveraging vector search for faster responses and improved operational efficiency." preview_image: /blog/case-study-nyris/preview.png social_preview_image: /blog/case-study-nyris/preview.png @@ -34,7 +34,7 @@ During his time at Amazon, Lukasson observed that search engines like Google oft In their quest for the perfect visual search provider, Nyris ultimately decided to develop their own solution. -## The Path to Vector-based Visual Search +## The Path to Vector-Based Visual Search Initially in 2015, the team explored traditional search algorithms based on key value SIFT (Scale Invariant Feature Transform) features to locate specific elements within images. However, they quickly realized that these methods were imprecise and unreliable. To address this, Nyris began experimenting with the first Convolutional Neural Networks (CNNs) to extract embeddings for vector search. diff --git a/qdrant-landing/content/blog/case-study-shakudo.md b/qdrant-landing/content/blog/case-study-shakudo.md new file mode 100644 index 000000000..bc84caec8 --- /dev/null +++ b/qdrant-landing/content/blog/case-study-shakudo.md @@ -0,0 +1,43 @@ +--- +draft: false +title: "Qdrant and Shakudo: Secure & Performant Vector Search in VPC Environments" +short_description: "Transforming customer service in finance and insurance with vector search-based retrieval." +description: "Implementing vector search for enterprise AI via Qdrant's Hybrid Cloud integration into Shakudo’s virtual private cloud." +preview_image: /blog/case-study-shakudo/preview.png +social_preview_image: /blog/case-study-shakudo/preview.png +date: 2024-09-23T00:02:00Z +author: Qdrant +featured: false +tags: + - Shakudo + - Vector Search +--- + +We are excited to announce that Qdrant has partnered with [Shakudo](https://www.shakudo.io/), bringing [Qdrant Hybrid Cloud](https://qdrant.tech/hybrid-cloud/) to Shakudo’s virtual private cloud (VPC) deployments. This collaboration allows Shakudo clients to seamlessly integrate Qdrant’s high-performance vector database as a managed service into their private infrastructure, ensuring data sovereignty, scalability, and low-latency vector search for enterprise AI applications. + +## Data Sovereignty and Compliance with Secure Vector Search + +Shakudo’s VPC deployments ensure that client data remains within their infrastructure, providing strict control over sensitive information while leveraging a fully managed AI toolset. Qdrant Hybrid Cloud is tailored for environments where data privacy and regulatory compliance are paramount. It keeps the data plane inside the customer's infrastructure, with only essential telemetry shared externally, guaranteeing database isolation and security, while providing a fully managed service. + +![shakudo-case-study](/blog/case-study-shakudo/shakudo-case-study.jpg) + +## Scaling and Performance Optimization for Enterprise Vector Search + +Qdrant Hybrid Cloud is optimized for Kubernetes, allowing for fast, automated deployments and hands-off cluster management. Shakudo’s platform, designed for VPC-based environments, allows businesses to deploy Qdrant’s vector search clusters with no DevOps overhead. Qdrant’s ability to handle billions of vectors - powered by our customized Hierarchical Navigable Small World (HNSW) indexing - ensures real-time processing and high accuracy for AI-driven applications like semantic search, recommendation systems, and retrieval-augmented generation (RAG). + +## Staying Compatible with the Entire Stack + +By deploying Qdrant Hybrid Cloud on Shakudo, organizations gain immediate compatibility with their existing data sources, pipelines, and applications. It integrates seamlessly with the existing stack, ensuring smooth and efficient operation across all components. As business needs evolve, the data stack can easily scale and adapt to new demands. + +## Key Benefits of Qdrant in Shakudo's Virtual Private Cloud + +- **Data Privacy & Control**: Shakudo users can run a Qdrant vector database inside their own VPC, ensuring sensitive data never leaves their infrastructure, while enjoying a managed service for simplicity and reliability. +- **Seamless Integration**: Qdrant’s Kubernetes-native setup allows rapid deployment on Shakudo’s VPC-based infrastructure, which provides pre-configured environments optimized for AI workloads. +- **Scalability**: Qdrant’s ability to handle billions of vectors and its high-performance indexing like HNSW make it ideal for applications requiring fast, accurate similarity searches. +- **Enterprise Flexibility**: With both on-premise and cloud-native setups available, this partnership offers businesses the flexibility to balance operational needs with privacy requirements​. + +## Learn More + +Ready to learn how Qdrant on Shakudo can enhance your AI infrastructure? Contact the Shakudo team to explore how they can help you deploy secure, high-performance vector search in your VPC environment, or get started [here](https://www.shakudo.io/integrations/qdrant). + +If you are interested in Qdrant’s Managed Cloud, Hybrid Cloud, or Private Cloud solutions for flexible deployment options for top-tier data privacy, [contact us](https://qdrant.tech/contact-sales/). diff --git a/qdrant-landing/content/blog/hybrid-cloud-airbyte.md b/qdrant-landing/content/blog/hybrid-cloud-airbyte.md index a3bd6cac7..d6915dae4 100644 --- a/qdrant-landing/content/blog/hybrid-cloud-airbyte.md +++ b/qdrant-landing/content/blog/hybrid-cloud-airbyte.md @@ -13,7 +13,7 @@ tags: - Vector Database --- -In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](/hybrid-cloud/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed vector database that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product. +In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](/hybrid-cloud/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed [vector database](/articles/what-is-a-vector-database/) that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product. This is a major step forward in offering enterprise customers incredible synergy for maximizing the potential of their AI data. Qdrant's new Kubernetes-native design, coupled with Airbyte’s powerful data ingestion pipelines meet the needs of developers who are both prototyping and building production-level apps. Airbyte simplifies the process of data integration by providing a platform that connects to various sources and destinations effortlessly. Moreover, Qdrant Hybrid Cloud leverages advanced indexing and search capabilities to empower users to explore and analyze their data efficiently. diff --git a/qdrant-landing/content/blog/hybrid-cloud-haystack.md b/qdrant-landing/content/blog/hybrid-cloud-haystack.md index 79579bab7..1a5797a39 100644 --- a/qdrant-landing/content/blog/hybrid-cloud-haystack.md +++ b/qdrant-landing/content/blog/hybrid-cloud-haystack.md @@ -13,7 +13,7 @@ tags: - Vector Database --- -We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering, allowing developers to deploy a managed vector database in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations. +We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering, allowing developers to deploy a managed [vector database](/articles/what-is-a-vector-database/) in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations. With a highly customizable framework like Haystack, implementing vector search becomes incredibly simple. Qdrant's new Qdrant Hybrid Cloud offering and its Kubernetes-native design supports customers all the way from a simple prototype setup to a production scenario on any hosting platform. Users can attach AI functionalities to their existing in-house software by creating custom integration components. Don’t forget, both products are open-source and highly modular! diff --git a/qdrant-landing/content/blog/qdrant-deeplearning-ai-course.md b/qdrant-landing/content/blog/qdrant-deeplearning-ai-course.md new file mode 100644 index 000000000..296f29b19 --- /dev/null +++ b/qdrant-landing/content/blog/qdrant-deeplearning-ai-course.md @@ -0,0 +1,54 @@ +--- +draft: false +title: "New DeepLearning.AI Course on Retrieval Optimization: From Tokenization to Vector Quantization" +short_description: "Free, beginner-friendly course to learn retrieval optimization and boost search performance." +description: "Join Qdrant and DeepLearning.AI’s free, beginner-friendly course to learn retrieval optimization and boost search performance in machine learning." +preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg +social_preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg +date: 2024-10-06T00:02:00Z +author: Qdrant +featured: false +tags: + - DeepLearning.AI + - Vector Search + - Vector Quantization + - Tokenization + - Retrieval-Augmented Generation + - Vector Database +--- + +We’re excited to announce a new course on DeepLearning.AI's platform: [Retrieval Optimization: From Tokenization to Vector Quantization](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo). This collaboration between Qdrant and DeepLearning.AI aims to empower developers and data enthusiasts with the skills needed to enhance [vector search](/advanced-search/) capabilities in their applications. + +Led by Qdrant’s Kacper Łukawski, this free, one-hour course is designed for beginners eager to delve into the world of retrieval optimization. + +## Why This Collaboration Matters + +At Qdrant, we believe in the power of effective search to transform user experiences. Partnering with DeepLearning.AI allows us to combine our cutting-edge vector search technology with their educational expertise, providing learners with a comprehensive understanding of how to build and optimize [Retrieval-Augmented Generation (RAG)](/rag/rag-evaluation-guide/) applications. This course is part of our commitment to equip the community with practical skills that leverage advanced machine learning techniques. + + + +## What You’ll Learn + +In this course, you’ll explore key concepts that will enhance your understanding of retrieval optimization: + +- Learn how tokenization works in large language and embedding models and how the tokenizer can affect the quality of your search. +- Explore how different tokenization techniques including Byte-Pair Encoding, WordPiece, and Unigram are trained and work. +- Understand how to [measure the quality of your retrieval](/rag/rag-evaluation-guide/) and how to optimize your search by adjusting HNSW parameters and [vector quantizations](/articles/what-is-vector-quantization/). + +## Who Should Enroll + +This course is tailored for anyone with basic Python knowledge. + +Whether you’re starting your journey in machine learning or looking to enhance your existing skills, this course offers valuable insights to boost your capabilities. + +### At a Glance: + +- **Speaker**: Kacper Łukawski, Qdrant Developer Advocate +- **Level**: Beginner +- **Cost**: Free +- **Location**: Online +- **Duration**: 1 Hour + +## How to Enroll + +[Enroll via the DeepLearning.AI website](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo). \ No newline at end of file diff --git a/qdrant-landing/content/blog/qdrant-for-startups-launch.md b/qdrant-landing/content/blog/qdrant-for-startups-launch.md new file mode 100644 index 000000000..fe4b97bd4 --- /dev/null +++ b/qdrant-landing/content/blog/qdrant-for-startups-launch.md @@ -0,0 +1,84 @@ +--- +draft: false +title: "Introducing Qdrant for Startups" +short_description: "Join our Startup Program now and scale your AI-driven applications with ease." +description: "Enjoy special discounts from Qdrant, HuggingFace, LlamaIndex, and Airbyte, as well as expert support & tooling perks, and be the first to try new features." +preview_image: /blog/qdrant-for-startups-launch/preview.png +social_preview_image: /blog/qdrant-for-startups-launch/preview.png +date: 2024-10-02T00:02:00Z +author: Qdrant +featured: false +tags: + - Qdrant + - Startups + - Vector Search + - Vector Database +--- + +# Supporting Early-Stage Startups + +Over the past few years, we’ve witnessed some of the most innovative AI applications being built on Qdrant. A significant number of these have come from startups pushing the boundaries of what’s possible in AI. To ensure these pioneering teams have access to the right resources at the right time, we're introducing **Qdrant for Startups**. This initiative is designed to provide startups with the technical support, guidance, and infrastructure they need to scale their AI innovations quickly and effectively. + +Qdrant for Startups helps early-stage startups fully leverage the capabilities of vector search technology. Whether you're building retrieval-augmented generation (RAG) systems, recommendation engines, or anomaly detection models, the program offers exclusive benefits, such as discounts for Qdrant cloud, expert technical guidance, exclusive partner benefits, and co-marketing opportunities - empowering you to build and scale your AI products efficiently and cost-effectively. + +## Benefits for admitted startups: + +- **Qdrant Cloud discount:** 20% discount on Qdrant Cloud valid for 12 months, optimizing costs while scaling with advanced vector search capabilities. +- **Expert technical guidance:** Dedicated technical support and guidance to optimize your application’s performance with vector search. +- **Co-marketing opportunities:** Collaboration with the Qdrant team on joint marketing initiatives to boost your startup’s visibility. +- **Early access to features:** Exclusive early access to upcoming Qdrant features, keeping you at the forefront of technological advancements. +- **Community access:** Access to Qdrant’s developer and AI community for collaboration, networking, and shared learning. + +## Access to popular AI tools + +We’ve built this program to support startups with their entire AI tech stack. In addition to Qdrant, accepted startups will receive exclusive discounts from our program partners - Hugging Face, LlamaIndex, and Airbyte - ensuring you have access to the key tools and resources needed to build and scale AI-driven applications. + +Accepted startup program members will have the ability to get additional benefits: + +- Hugging Face: $100 compute credits for the HuggingFace Hub +- LlamaIndex: 20% discount for 12 months for LlamaCloud +- Airbyte: Cloud credits for Y Combinator startups + +[![qdrant-for-startups-launch](/blog/qdrant-for-startups-launch/startup-cta.png)](https://qdrant.tech/qdrant-for-startups) + +## Frequently Asked Questions: + +**Q: What are the eligibility requirements?** + +A: You must meet all of the following: + +- Pre-seed, Seed or Series A startups (under five years old) +- New user of Qdrant Cloud +- Has not previously participated in the Qdrant for Startups program +- Offer is not valid to existing Qdrant customers +- Must be building an AI-driven product or services (agencies or devshops are not eligible) +- A live, functional website is required +- Billing must be done directly with Qdrant (not through a marketplace) + +**Q: How can I apply to the Qdrant Startup Program?** + +A: Apply through our online form by providing details about your startup and its plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities. + +**Q: What criteria are used to select startups for the program?** + +A: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected. + +**Q: How long is the discount valid, and are there any conditions?** + +A: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount. + +**Q: How can I maximize the co-marketing opportunities offered by the program?** + +A: Engage actively with our marketing team for features on social media, possible appearances in Discord talks or webinars, and case studies to maximize your startup's visibility and showcase your innovative use of Qdrant. + +**Q: Can existing Qdrant customers apply for the Startup Program?** + +A: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered by the program. + +**Q: Can I reapply if my application is initially rejected?** + +A: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply. + +**Q: Who can I contact for more information about the program?** + +A: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com. diff --git a/qdrant-landing/content/blog/qdrant-stars-announcement copy.md b/qdrant-landing/content/blog/qdrant-stars-announcement copy.md index 1328d213a..ca447c149 100644 --- a/qdrant-landing/content/blog/qdrant-stars-announcement copy.md +++ b/qdrant-landing/content/blog/qdrant-stars-announcement copy.md @@ -130,7 +130,7 @@ I'm really excited to show the power of the Qdrant as vector database. Especiall We are happy to welcome this group of people who are deeply committed to advancing vector search technology. We look forward to supporting their vision, and helping them make a bigger impact on the community. -You can find and chat with them at our [Discord Community](discord.gg/qdrant). +You can find and chat with them at our [Discord Community](https://discord.gg/qdrant/). ### Why become a Qdrant Star? diff --git a/qdrant-landing/content/blog/virtualbrain-best-rag-to-unleash-the-real-power-of-ai-guillaume-marquis-vector-space-talks.md b/qdrant-landing/content/blog/virtualbrain-best-rag-to-unleash-the-real-power-of-ai-guillaume-marquis-vector-space-talks.md index 6bca2548b..c6f2cf8ae 100644 --- a/qdrant-landing/content/blog/virtualbrain-best-rag-to-unleash-the-real-power-of-ai-guillaume-marquis-vector-space-talks.md +++ b/qdrant-landing/content/blog/virtualbrain-best-rag-to-unleash-the-real-power-of-ai-guillaume-marquis-vector-space-talks.md @@ -35,7 +35,7 @@ Guillaume Marquis, a dedicated Engineer and AI enthusiast, serves as the Chief T Who knew that document retrieval could be creative? Guillaume and VirtualBrain help draft sales proposals using past reports. It's fascinating how tech aids deep work beyond basic search tasks. -Tackling document retrieval and AI assistance, Guillaume furthermore unpacks the ins and outs of searching through vast data using a scoring system, the virtue of RAG for deep work, and going through the 'illusion of work', enhancing insights for knowledge workers while confronting the challenges of scalability and user feedback on hallucinations. +Tackling document retrieval and AI assistance, Guillaume furthermore unpacks the ins and outs of searching through vast data using a scoring system, the virtue of [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) for deep work, and going through the 'illusion of work', enhancing insights for knowledge workers while confronting the challenges of scalability and user feedback on hallucinations. Here are some key insight from this episode you need to look out for: @@ -305,7 +305,7 @@ Guillaume Marquis: So you can trade for free. Demetrios: -Even better. Look at that, Christmas came early. Well, let's go have some fun, play around with it. And I can't promise, but I may give you some feedback, I may give you some evaluation metrics if it's hallucinating. +Even better. Look at that, Christmas came early. Well, let's go have some fun, play around with it. And I can't promise, but I may give you some feedback, I may give you some [evaluation](https://qdrant.tech/rag/rag-evaluation-guide/) metrics if it's hallucinating. Guillaume Marquis: Or what if I see some thumbs up or thumbs down, I will know that it's you. diff --git a/qdrant-landing/content/documentation/3-dl.md b/qdrant-landing/content/documentation/3-dl.md index 7743e102e..7405aefb5 100644 --- a/qdrant-landing/content/documentation/3-dl.md +++ b/qdrant-landing/content/documentation/3-dl.md @@ -2,7 +2,7 @@ #Delimiter files are used to separate the list of documentation pages into sections. title: "Examples" type: delimiter -weight: 23 # Change this weight to change order of sections +weight: 24 # Change this weight to change order of sections sitemapExclude: True _build: publishResources: false diff --git a/qdrant-landing/content/documentation/5-dl.md b/qdrant-landing/content/documentation/5-dl.md index d70c81146..98a9775b2 100644 --- a/qdrant-landing/content/documentation/5-dl.md +++ b/qdrant-landing/content/documentation/5-dl.md @@ -2,7 +2,7 @@ #Delimiter files are used to separate the list of documentation pages into sections. title: "Support" type: delimiter -weight: 26 # Change this weight to change order of sections +weight: 27 # Change this weight to change order of sections sitemapExclude: True _build: publishResources: false diff --git a/qdrant-landing/content/documentation/cloud/backups.md b/qdrant-landing/content/documentation/cloud/backups.md index 2fb6ca4c5..ac461df72 100644 --- a/qdrant-landing/content/documentation/cloud/backups.md +++ b/qdrant-landing/content/documentation/cloud/backups.md @@ -82,11 +82,11 @@ Here is how you can take a snapshot and recover a collection: - For a single node cluster, call the snapshot endpoint on the exposed URL. - For a multi node cluster call a snapshot on each node of the collection. Specifically, prepend `node-{num}-` to your cluster URL. - Then call the [snapshot endpoint](../../concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0. + Then call the [snapshot endpoint](/documentation/concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0. - In the response, you'll see the name of the snapshot. 2. Delete and recreate the collection. 3. Recover the snapshot: - - Call the [recover endpoint](../../concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host. + - Call the [recover endpoint](/documentation/concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host. ## Backup considerations diff --git a/qdrant-landing/content/documentation/cloud/capacity-sizing.md b/qdrant-landing/content/documentation/cloud/capacity-sizing.md deleted file mode 100644 index 092b657a9..000000000 --- a/qdrant-landing/content/documentation/cloud/capacity-sizing.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -title: Configure Size & Capacity -weight: 40 -aliases: - - capacity ---- - -# Configuring Qdrant Cloud Cluster Capacity and Size - -We have been asked a lot about the optimal cluster configuration to serve a number of vectors. -The only right answer is “It depends”. - -It depends on a number of factors and options you can choose for your collections. - -## Basic configuration - -If you need to keep all vectors in memory for maximum performance, there is a very rough formula for estimating the needed memory size looks like this: - -```text -memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5 -``` - -Extra 50% is needed for metadata (indexes, point versions, etc.) as well as for temporary segments constructed during the optimization process. - -If you need to have payloads along with the vectors, it is recommended to store it on the disc, and only keep [indexed fields](../../concepts/indexing/#payload-index) in RAM. -Read more about the payload storage in the [Storage](../../concepts/storage/#payload-storage) section. - - -## Storage focused configuration - -If your priority is to serve large amount of vectors with an average search latency, it is recommended to configure [mmap storage](../../concepts/storage/#configuring-memmap-storage). -In this case vectors will be stored on the disc in memory-mapped files, and only the most frequently used vectors will be kept in RAM. - -The amount of available RAM will significantly affect the performance of the search. -As a rule of thumb, if you keep 2 times less vectors in RAM, the search latency will be 2 times lower. - -The speed of disks is also important. [Let us know](/documentation/support/) if you have special requirements for a high-volume search. - -## Sub-groups oriented configuration - - -If your use case assumes that the vectors are split into multiple collections or sub-groups based on payload values, -it is recommended to configure memory-map storage. -For example, if you serve search for multiple users, but each of them has an subset of vectors which they use independently. - -In this scenario only the active subset of vectors will be kept in RAM, which allows -the fast search for the most active and recent users. - -In this case you can estimate required memory size as follows: - -```text -memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5 -``` - -## Disk space - -Clusters that support vector search require significant disk space. If you're -running low on disk space in your cluster, you can use the UI at -[cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster. - - - -If you're running low on disk space, consider the following advantages: - -- Larger Datasets: Supports larger datasets. With vector search, -larger datasets can improve the relevance and quality of search results. -- Improved Indexing: Supports the use of indexing strategies such as -HNSW (Hierarchical Navigable Small World). -- Caching: Improves speed when you cache frequently accessed data on disk. -- Backups and Redundancy: Allows more frequent backups. Perhaps the most important advantage. diff --git a/qdrant-landing/content/documentation/cloud/cluster-scaling.md b/qdrant-landing/content/documentation/cloud/cluster-scaling.md index 44b928423..09d230eda 100644 --- a/qdrant-landing/content/documentation/cloud/cluster-scaling.md +++ b/qdrant-landing/content/documentation/cloud/cluster-scaling.md @@ -27,11 +27,11 @@ Vertical scaling can be an effective way to improve the performance of a cluster In such cases, horizontal scaling may be a more effective solution. -Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](../../guides/distributed_deployment/#sharding) section for details. +Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](/documentation/guides/distributed_deployment/#sharding) section for details. After that, you can configure, or change the amount of Qdrant database nodes within a cluster during cluster creation, or on the cluster detail page via "Scale" button. -Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](../../guides/distributed_deployment/#cluster-scaling). +Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](/documentation/guides/distributed_deployment/#cluster-scaling). Note, that it is currently not possible to horizontally scale down the cluster in the Qdrant Cloud UI. If you require a horizontal scale down, please open a support ticket. diff --git a/qdrant-landing/content/documentation/cloud/create-cluster.md b/qdrant-landing/content/documentation/cloud/create-cluster.md index 115486fd9..db04f8b60 100644 --- a/qdrant-landing/content/documentation/cloud/create-cluster.md +++ b/qdrant-landing/content/documentation/cloud/create-cluster.md @@ -20,7 +20,7 @@ A free tier cluster only includes 1 single node with the following resources: | Disk space | 4 GB | | Nodes | 1 | -This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity and sizing](/documentation/cloud/capacity-sizing/). +This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity Planning](/documentation/guides/capacity-planning/). The choice of cloud providers and regions is limited. @@ -73,7 +73,7 @@ This page shows you how to use the Qdrant Cloud Console to create a custom Qdran 1. Choose your data center region or Hybrid Cloud environment. 1. Configure RAM for each node. - > For more information, see our [**Capacity and Sizing**](/documentation/cloud/capacity-sizing/) guidance. + > For more information, see our [Capacity Planning](/documentation/guides/capacity-planning/) guidance. 1. Choose the number of vCPUs per node. If you add more RAM, the menu provides different options for vCPUs. 1. Select the number of nodes you want the cluster to be deployed on. diff --git a/qdrant-landing/content/documentation/concepts/collections.md b/qdrant-landing/content/documentation/concepts/collections.md index ddc951979..548aa6020 100644 --- a/qdrant-landing/content/documentation/concepts/collections.md +++ b/qdrant-landing/content/documentation/concepts/collections.md @@ -28,7 +28,7 @@ These settings can be changed at any time by a corresponding request. ## Setting up multitenancy -**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](../../tutorials/multiple-partitions/) +**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](/documentation/tutorials/multiple-partitions/) **When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise. @@ -139,12 +139,12 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ In addition to the required options, you can also specify custom values for the following collection options: -* `hnsw_config` - see [indexing](../indexing/#vector-index) for details. -* `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](../storage/#versioning) -* `optimizers_config` - see [optimizer](../optimizer/) for details. -* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](../../guides/distributed_deployment/#sharding) section for details. +* `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details. +* `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](/documentation/concepts/storage/#versioning) +* `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details. +* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](/documentation/guides/distributed_deployment/#sharding) section for details. * `on_disk_payload` - defines where to store payload data. If `true` - payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload. -* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details. +* `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details. Default parameters for the optional collection parameters are defined in [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml). @@ -155,7 +155,7 @@ See [schema definitions](https://api.qdrant.tech/api-reference/collections/creat Vectors all live in RAM for very quick access. The `on_disk` parameter can be set in the vector configuration. If true, all vectors will live on disk. This will enable the use of -[memmaps](../../concepts/storage/#configuring-memmap-storage), +[memmaps](/documentation/concepts/storage/#configuring-memmap-storage), which is suitable for ingesting a large amount of data. ### Create collection from another collection @@ -466,8 +466,8 @@ For rare use cases, it is possible to create a collection without any vector sto *Available as of v1.1.1* For each named vector you can optionally specify -[`hnsw_config`](../indexing/#vector-index) or -[`quantization_config`](../../guides/quantization/#setting-up-quantization-in-qdrant) to +[`hnsw_config`](/documentation/concepts/indexing/#vector-index) or +[`quantization_config`](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) to deviate from the collection configuration. This can be useful to fine-tune search performance on a vector level. @@ -476,7 +476,7 @@ search performance on a vector level. Vectors all live in RAM for very quick access. On a per-vector basis you can set `on_disk` to true to store all vectors on disk at all times. This will enable the use of -[memmaps](../../concepts/storage/#configuring-memmap-storage), +[memmaps](/documentation/concepts/storage/#configuring-memmap-storage), which is suitable for ingesting a large amount of data. @@ -752,7 +752,7 @@ Outside of a unique name, there are no required configuration parameters for spa The distance function for sparse vectors is always `Dot` and does not need to be specified. -However, there are optional parameters to tune the underlying [sparse vector index](../indexing/#sparse-vector-index). +However, there are optional parameters to tune the underlying [sparse vector index](/documentation/concepts/indexing/#sparse-vector-index). ### Check collection existence @@ -928,9 +928,9 @@ client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ The following parameters can be updated: -* `optimizers_config` - see [optimizer](../optimizer/) for details. -* `hnsw_config` - see [indexing](../indexing/#vector-index) for details. -* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details. +* `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details. +* `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details. +* `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details. * `vectors` - vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings. * `params` - other collection parameters, including `write_consistency_factor` and `on_disk_payload`. @@ -1495,14 +1495,14 @@ round of automatic optimizations has completed. To clarify: these numbers don't represent the exact amount of points or vectors you have inserted, nor does it represent the exact number of distinguishable points or vectors you can query. If you want to know exact counts, refer to the -[count API](../points/#counting-points). +[count API](/documentation/concepts/points/#counting-points). _Note: these numbers may be removed in a future version of Qdrant._ ### Indexing vectors in HNSW In some cases, you might be surprised the value of `indexed_vectors_count` is lower than `vectors_count`. This is an intended behaviour and -depends on the [optimizer configuration](../optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the +depends on the [optimizer configuration](/documentation/concepts/optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the value of `indexing_threshold`(in kB). If your collection is very small or the dimensionality of the vectors is low, there might be no HNSW segment created and `indexed_vectors_count` might be equal to `0`. diff --git a/qdrant-landing/content/documentation/concepts/explore.md b/qdrant-landing/content/documentation/concepts/explore.md index 797c472fd..9dedae023 100644 --- a/qdrant-landing/content/documentation/concepts/explore.md +++ b/qdrant-landing/content/documentation/concepts/explore.md @@ -7,7 +7,7 @@ aliases: # Explore the data -After mastering the concepts in [search](../search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning. +After mastering the concepts in [search](/documentation/concepts/search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning. ## Recommendation API diff --git a/qdrant-landing/content/documentation/concepts/filtering.md b/qdrant-landing/content/documentation/concepts/filtering.md index 402709d7f..8521ac246 100644 --- a/qdrant-landing/content/documentation/concepts/filtering.md +++ b/qdrant-landing/content/documentation/concepts/filtering.md @@ -8,7 +8,7 @@ aliases: # Filtering With Qdrant, you can set conditions when searching or retrieving points. -For example, you can impose conditions on both the [payload](../payload/) and the `id` of the point. +For example, you can impose conditions on both the [payload](/documentation/concepts/payload/) and the `id` of the point. Setting additional conditions is important when it is impossible to express all the features of the object in the embedding. Examples include a variety of business requirements: stock availability, user location, or desired price range. @@ -838,7 +838,7 @@ qdrant.NewMatchInt("count", 0) The simplest kind of condition is one that checks if the stored value equals the given one. If several values are stored, at least one of them should match the condition. -You can apply it to [keyword](../payload/#keyword), [integer](../payload/#integer) and [bool](../payload/#bool) payloads. +You can apply it to [keyword](/documentation/concepts/payload/#keyword), [integer](/documentation/concepts/payload/#integer) and [bool](/documentation/concepts/payload/#bool) payloads. ### Match Any @@ -847,7 +847,7 @@ You can apply it to [keyword](../payload/#keyword), [integer](../payload/#intege In case you want to check if the stored value is one of multiple values, you can use the Match Any condition. Match Any works as a logical OR for the given values. It can also be described as a `IN` operator. -You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads. +You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads. Example: @@ -909,7 +909,7 @@ In case you want to check if the stored value is not one of multiple values, you Match Except works as a logical NOR for the given values. It can also be described as a `NOT IN` operator. -You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads. +You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads. Example: @@ -1908,7 +1908,7 @@ A special case of the `match` condition is the `text` match condition. It allows you to search for a specific substring, token or phrase within the text field. Exact texts that will match the condition depend on full-text index configuration. -Configuration is defined during the index creation and describe at [full-text index](../indexing/#full-text-index). +Configuration is defined during the index creation and describe at [full-text index](/documentation/concepts/indexing/#full-text-index). If there is no full-text index for the field, the condition will work as exact substring match. @@ -2047,11 +2047,11 @@ Comparisons that can be used: - `lt` - less than - `lte` - less than or equal -Can be applied to [float](../payload/#float) and [integer](../payload/#integer) payloads. +Can be applied to [float](/documentation/concepts/payload/#float) and [integer](/documentation/concepts/payload/#integer) payloads. ### Datetime Range -The datetime range is a unique range condition, used for [datetime](../payload/#datetime) payloads, which supports RFC 3339 formats. +The datetime range is a unique range condition, used for [datetime](/documentation/concepts/payload/#datetime) payloads, which supports RFC 3339 formats. You do not need to convert dates to UNIX timestaps. During comparison, timestamps are parsed and converted to UTC. _Available as of v1.8.0_ @@ -2364,7 +2364,7 @@ qdrant.NewGeoRadius("location", 52.520711, 13.403683, 1000.0) It matches with `location`s inside a circle with the `center` at the center and a radius of `radius` meters. If several values are stored, at least one of them should match the condition. -These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo). +These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo). #### Geo Polygon Geo Polygons search is useful for when you want to find points inside an irregularly shaped area, for example a country boundary or a forest boundary. A polygon always has an exterior ring and may optionally include interior rings. A lake with an island would be an example of an interior ring. If you wanted to find points in the water but not on the island, you would make an interior ring for the island. @@ -2659,7 +2659,7 @@ qdrant.NewGeoPolygon("location", A match is considered any point location inside or on the boundaries of the given polygon's exterior but not inside any interiors. If several location values are stored for a point, then any of them matching will include that point as a candidate in the resultset. -These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo). +These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo). ### Values count diff --git a/qdrant-landing/content/documentation/concepts/hybrid-queries.md b/qdrant-landing/content/documentation/concepts/hybrid-queries.md index cd5bb2f18..3935ecd13 100644 --- a/qdrant-landing/content/documentation/concepts/hybrid-queries.md +++ b/qdrant-landing/content/documentation/concepts/hybrid-queries.md @@ -10,7 +10,7 @@ hideInSidebar: false # Optional. If true, the page will not be shown in the side *Available as of v1.10.0* -With the introduction of [many named vectors per point](../vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries, +With the introduction of [many named vectors per point](/documentation/concepts/vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries, or by performing the search in more than one stage. Qdrant has a flexible and universal interface to make this possible, called `Query API` ([API reference](https://api.qdrant.tech/api-reference/search/query-points)). @@ -793,7 +793,7 @@ Other than the introduction of `prefetch`, the `Query API` has been designed to ### Query by ID -Whenever you need to use a vector as an input, you can always use a [point ID](../points/#point-ids) instead. +Whenever you need to use a vector as an input, you can always use a [point ID](/documentation/concepts/points/#point-ids) instead. ```http POST /collections/{collection_name}/points/query @@ -1397,4 +1397,4 @@ client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{ }) ``` -For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](./search/#search-groups) and [lookup](./search/#lookup-in-groups). +For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](/documentation/concepts/search/#search-groups) and [lookup](/documentation/concepts/search/#lookup-in-groups). diff --git a/qdrant-landing/content/documentation/concepts/indexing.md b/qdrant-landing/content/documentation/concepts/indexing.md index d1691116a..121bbe20c 100644 --- a/qdrant-landing/content/documentation/concepts/indexing.md +++ b/qdrant-landing/content/documentation/concepts/indexing.md @@ -12,14 +12,14 @@ A key feature of Qdrant is the effective combination of vector and traditional i The indexes in the segments exist independently, but the parameters of the indexes themselves are configured for the whole collection. Not all segments automatically have indexes. -Their necessity is determined by the [optimizer](../optimizer/) settings and depends, as a rule, on the number of stored points. +Their necessity is determined by the [optimizer](/documentation/concepts/optimizer/) settings and depends, as a rule, on the number of stored points. ## Payload Index Payload index in Qdrant is similar to the index in conventional document-oriented databases. This index is built for a specific field and type, and is used for quick point requests by the corresponding filtering condition. -The index is also used to accurately estimate the filter cardinality, which helps the [query planning](../search/#query-planning) choose a search strategy. +The index is also used to accurately estimate the filter cardinality, which helps the [query planning](/documentation/concepts/search/#query-planning) choose a search strategy. Creating an index requires additional computational resources and memory, so choosing fields to be indexed is essential. Qdrant does not make this choice but grants it to the user. @@ -119,19 +119,19 @@ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection }) ``` -You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](../filtering/#nested-key). +You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](/documentation/concepts/filtering/#nested-key). Available field types are: -* `keyword` - for [keyword](../payload/#keyword) payload, affects [Match](../filtering/#match) filtering conditions. -* `integer` - for [integer](../payload/#integer) payload, affects [Match](../filtering/#match) and [Range](../filtering/#range) filtering conditions. -* `float` - for [float](../payload/#float) payload, affects [Range](../filtering/#range) filtering conditions. -* `bool` - for [bool](../payload/#bool) payload, affects [Match](../filtering/#match) filtering conditions (available as of v1.4.0). -* `geo` - for [geo](../payload/#geo) payload, affects [Geo Bounding Box](../filtering/#geo-bounding-box) and [Geo Radius](../filtering/#geo-radius) filtering conditions. -* `datetime` - for [datetime](../payload/#datetime) payload, affects [Range](../filtering/#range) filtering conditions (available as of v1.8.0). -* `text` - a special kind of index, available for [keyword](../payload/#keyword) / string payloads, affects [Full Text search](../filtering/#full-text-match) filtering conditions. -* `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](../payload/#uuid). -Affects [Match](../filtering/#match) filtering conditions. (available as of v1.11.0) +* `keyword` - for [keyword](/documentation/concepts/payload/#keyword) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions. +* `integer` - for [integer](/documentation/concepts/payload/#integer) payload, affects [Match](/documentation/concepts/filtering/#match) and [Range](/documentation/concepts/filtering/#range) filtering conditions. +* `float` - for [float](/documentation/concepts/payload/#float) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions. +* `bool` - for [bool](/documentation/concepts/payload/#bool) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions (available as of v1.4.0). +* `geo` - for [geo](/documentation/concepts/payload/#geo) payload, affects [Geo Bounding Box](/documentation/concepts/filtering/#geo-bounding-box) and [Geo Radius](/documentation/concepts/filtering/#geo-radius) filtering conditions. +* `datetime` - for [datetime](/documentation/concepts/payload/#datetime) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions (available as of v1.8.0). +* `text` - a special kind of index, available for [keyword](/documentation/concepts/payload/#keyword) / string payloads, affects [Full Text search](/documentation/concepts/filtering/#full-text-match) filtering conditions. +* `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](/documentation/concepts/payload/#uuid). +Affects [Match](/documentation/concepts/filtering/#match) filtering conditions. (available as of v1.11.0) Payload index may occupy some additional memory, so it is recommended to only use index for those fields that are used in filtering conditions. If you need to filter by many fields and the memory limits does not allow to index all of them, it is recommended to choose the field that limits the search result the most. @@ -313,7 +313,7 @@ Available tokenizers are: * `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`. * `multilingual` - special type of tokenizer based on [charabia](https://github.com/meilisearch/charabia) package. It allows proper tokenization and lemmatization for multiple languages, including those with non-latin alphabets and non-space delimiters. See [charabia documentation](https://github.com/meilisearch/charabia) for full list of supported languages supported normalization options. In the default build configuration, qdrant does not include support for all languages, due to the increasing size of the resulting binary. Chinese, Japanese and Korean languages are not enabled by default, but can be enabled by building qdrant from source with `--features multiling-chinese,multiling-japanese,multiling-korean` flags. -See [Full Text match](../filtering/#full-text-match) for examples of querying with full-text index. +See [Full Text match](/documentation/concepts/filtering/#full-text-match) for examples of querying with full-text index. ### Parameterized index @@ -645,7 +645,7 @@ The list will be extended in future versions. Many vector search use-cases require multitenancy. In a multi-tenant scenario the collection is expected to contain multiple subsets of data, where each subset belongs to a different tenant. -Qdrant supports efficient multi-tenant search by enabling [special configuration](../guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant. +Qdrant supports efficient multi-tenant search by enabling [special configuration](/documentation/guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant.