mirror of
https://github.com/qdrant/landing_page.git
synced 2026-09-28 15:38:33 +02:00
Merge branch 'master' into devportal-blog-1
This commit is contained in:
@@ -27,11 +27,11 @@ jobs:
|
||||
export PATH="${CURRENT_DIR}/dart-sass:${PATH}"
|
||||
cd qdrant-landing && hugo --gc -b 'http://localhost:1313' && hugo serve &
|
||||
sleep 5 # wait for server to start
|
||||
- name: Link Checker
|
||||
- name: Internal Links Check
|
||||
id: lychee
|
||||
uses: lycheeverse/lychee-action@v1.8.0
|
||||
with:
|
||||
args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' qdrant-landing/public/
|
||||
args: --max-redirects 0 --exclude '.*' --include 'http://localhost:1313/.*' --base http://localhost:1313/ qdrant-landing/public/
|
||||
fail: true
|
||||
env:
|
||||
GITHUB_TOKEN: ${{secrets.GITHUB_TOKEN}}
|
||||
|
||||
@@ -101,207 +101,6 @@ disableKinds = ["taxonomy", "term"]
|
||||
category = "categories"
|
||||
example = "examples"
|
||||
|
||||
[menu]
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "product"
|
||||
name = "Product"
|
||||
weight = 1
|
||||
[menu.main.params]
|
||||
in_header = true
|
||||
in_footer = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "use-case"
|
||||
name = "Use cases"
|
||||
weight = 1
|
||||
parent = "product"
|
||||
url = "/use-cases/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "solutions"
|
||||
name = "Solutions"
|
||||
weight = 2
|
||||
parent = "product"
|
||||
url = "/solutions/"
|
||||
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "benchmarks"
|
||||
name = "Benchmarks"
|
||||
weight = 3
|
||||
parent = "product"
|
||||
url = "/benchmarks/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "demo"
|
||||
name = "Demos"
|
||||
weight = 4
|
||||
parent = "product"
|
||||
url = "/demo/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "pricing"
|
||||
name = "Pricing"
|
||||
weight = 5
|
||||
parent = "product"
|
||||
url = "/pricing/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "resources"
|
||||
name = "Resources"
|
||||
weight = 2
|
||||
[menu.main.params]
|
||||
in_header = true
|
||||
in_footer = false
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "documentation"
|
||||
name = "Documentation"
|
||||
parent = "resources"
|
||||
weight = 1
|
||||
url = "/documentation/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "articles"
|
||||
name = "Articles"
|
||||
weight = 4
|
||||
parent = "resources"
|
||||
url = "/articles/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "blog"
|
||||
name = "Blog"
|
||||
weight = 5
|
||||
parent = "resources"
|
||||
url = "/blog/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "roadmap"
|
||||
name = "Roadmap"
|
||||
weight = 6
|
||||
parent = "resources"
|
||||
url = "https://qdrant.to/roadmap"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "changelog"
|
||||
name = "Changelog"
|
||||
weight = 7
|
||||
parent = "resources"
|
||||
url = "https://github.com/qdrant/qdrant/releases"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "trust-center"
|
||||
name = "Trust Center"
|
||||
weight = 8
|
||||
parent = "resources"
|
||||
url = "http://qdrant.to/trust-center"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "community"
|
||||
name = "Community"
|
||||
weight = 4
|
||||
[menu.main.params]
|
||||
in_header = true
|
||||
in_footer = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "github"
|
||||
name = "Github"
|
||||
weight = 1
|
||||
parent = "community"
|
||||
url = "https://github.com/qdrant/qdrant"
|
||||
pre = "<i class='fab fa-github'></i>"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "discord"
|
||||
name = "Discord"
|
||||
weight = 2
|
||||
parent = "community"
|
||||
url = "https://qdrant.to/discord"
|
||||
pre = "<i class='fab fa-discord'></i>"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "twitter"
|
||||
name = "Twitter"
|
||||
weight = 3
|
||||
parent = "community"
|
||||
url = "https://qdrant.to/twitter"
|
||||
pre = "<i class='fab fa-twitter'></i>"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "newsletter"
|
||||
name = "Newsletter"
|
||||
weight = 4
|
||||
parent = "community"
|
||||
url = "/subscribe/"
|
||||
pre = "<i class='fas fa-mail-bulk'></i>"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "contact"
|
||||
name = "Contact us"
|
||||
weight = 5
|
||||
parent = "community"
|
||||
url = "https://qdrant.to/contact-us"
|
||||
pre = "<i class='fas fa-envelope'></i>"
|
||||
[menu.main.params]
|
||||
external = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "company"
|
||||
name = "Company"
|
||||
weight = 4
|
||||
[menu.main.params]
|
||||
in_header = false
|
||||
in_footer = true
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "jobs"
|
||||
name = "Jobs"
|
||||
weight = 1
|
||||
parent = "company"
|
||||
url = "https://qdrant.join.com"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "privacy-policy"
|
||||
name = "Privacy Policy"
|
||||
weight = 2
|
||||
parent = "company"
|
||||
url = "/legal/privacy-policy/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "terms"
|
||||
name = "Terms"
|
||||
weight = 3
|
||||
parent = "company"
|
||||
url = "/legal/terms_and_conditions/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "impressum"
|
||||
name = "Impressum"
|
||||
weight = 4
|
||||
parent = "company"
|
||||
url = "/legal/impressum/"
|
||||
|
||||
[[menu.main]]
|
||||
identifier = "credits"
|
||||
name = "Credits"
|
||||
weight = 5
|
||||
parent = "company"
|
||||
url = "/legal/credits/"
|
||||
|
||||
[markup]
|
||||
[markup.goldmark.renderer]
|
||||
unsafe=true
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
---
|
||||
title: Qdrant Summer of Code 2024 - WASM based Dimension Reduction
|
||||
short_description: QSOC'24 WASM based Dimension Reduction
|
||||
description: My journey as a Qdrant Summer of Code 2024 participant working on enhancing vector visualization using WebAssembly (WASM) based dimension reduction.
|
||||
preview_dir: /articles_data/dimension-reduction-qsoc/preview
|
||||
small_preview_image: /articles_data/dimension-reduction-qsoc/icon.svg
|
||||
social_preview_image: /articles_data/dimension-reduction-qsoc/preview/social_preview.jpg
|
||||
weight: -10
|
||||
author: Jishan Bhattacharya
|
||||
author_link: https://www.linkedin.com/in/j16n/
|
||||
date: 2024-08-31T10:39:48.312Z
|
||||
draft: false
|
||||
keywords:
|
||||
|
||||
- dimension reduction
|
||||
- web assembly
|
||||
- qsoc'24
|
||||
- vector similarity
|
||||
- tsne
|
||||
- qdrant data visualization
|
||||
---
|
||||
|
||||
|
||||
|
||||
## Introduction
|
||||
|
||||
Hello, everyone! I'm Jishan Bhattacharya, and I had the incredible opportunity to intern at Qdrant this summer as part of the Qdrant Summer of Code 2024. Under the mentorship of [Andrey Vasnetsov](https://www.linkedin.com/in/andrey-vasnetsov-75268897/), I dived into the world of performance optimization, focusing on enhancing vector visualization using WebAssembly (WASM). In this article, I'll share the insights, challenges, and accomplishments from my journey — one filled with learning, experimentation, and plenty of coding adventures.
|
||||
|
||||
|
||||
## Project Overview
|
||||
|
||||
Qdrant is a robust vector database and search engine designed to store vector data and perform tasks like similarity search and clustering. One of its standout features is the ability to visualize high-dimensional vectors in a 2D space. However, the existing implementation faced performance bottlenecks, especially when scaling to large datasets. My mission was to tackle this challenge by leveraging a WASM-based solution for dimensionality reduction in the visualization process.
|
||||
|
||||
|
||||
## Learnings & Challenges
|
||||
|
||||
Our weapon of choice was Rust, paired with WASM, and we employed the t-SNE algorithm for dimensionality reduction. For those unfamiliar, t-SNE (t-Distributed Stochastic Neighbor Embedding) is a technique that helps visualize high-dimensional data by projecting it into two or three dimensions. It operates in two main steps:
|
||||
|
||||
1. **Computing Pairwise Similarity:** This step involves calculating the similarity between each pair of data points in the original high-dimensional space.
|
||||
|
||||
2. **Iterative Optimization:** The second step is iterative, where the embedding is refined using gradient descent. Here, the similarity matrix from the first step plays a crucial role.
|
||||
|
||||
At the outset, Andrey tasked me with rewriting the existing JavaScript implementation of t-SNE in Rust, introducing multi-threading along the way. Setting up WASM with Vite for multi-threaded execution was no small feat, but the effort paid off. The resulting Rust implementation outperformed the single-threaded JavaScript version, although it still struggled with large datasets.
|
||||
|
||||
Next came the challenge of optimizing the algorithm further. A key aspect of t-SNE's first step is finding the nearest neighbors for each data point, which requires an efficient data structure. I opted for a [Vantage Point Tree](https://en.wikipedia.org/wiki/Vantage-point_tree) (also known as a Ball Tree) to speed up this process. As for the second step, while it is inherently sequential, there was still room for improvement. I incorporated Barnes-Hut approximation to accelerate the gradient calculation. This method approximates the forces between points in low dimensional space, making the process more efficient.
|
||||
|
||||
To illustrate, imagine dividing a 2D space into quadrants, each containing multiple points. Every quadrant is again subdivided into four quadrants. This is done until every point belongs to a single cell.
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/barnes_hut.png"
|
||||
caption="Barnes-Hut Approximation"
|
||||
alt="Calculating the resultant force on red point using Barnes-Hut approximation"
|
||||
>}}
|
||||
|
||||
We then calculate the center of mass for each cell represented by a blue circle as shown in the figure. Now let’s say we want to find all the forces, represented by dotted lines, on the red point. Barnes Hut’s approximation states that for points that are sufficiently distant, instead of computing the force for each individual point, we use the center of mass as a proxy, significantly reducing the computational load. This is represented by the blue dotted line in the figure.
|
||||
|
||||
These optimizations made a remarkable difference — Barnes-Hut t-SNE was eight times faster than the exact t-SNE for 10,000 vectors.
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/rust_rewrite.jpg"
|
||||
caption="Exact t-SNE - Total time: 884.728s"
|
||||
alt="Image of visualizing 10,000 vectors using exact t-SNE which took 884.728s"
|
||||
>}}
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/rust_bhtsne.jpg"
|
||||
caption="Barnes-Hut t-SNE - Total time: 104.191s"
|
||||
alt="Image of visualizing 10,000 vectors using Barnes-Hut t-SNE which took 110.728s"
|
||||
>}}
|
||||
|
||||
Despite these improvements, the first step of the algorithm was still a bottleneck, leading to noticeable delays and blank screens. I experimented with approximate nearest neighbor algorithms, but the performance gains were minimal. After consulting with my mentor, we decided to compute the nearest neighbors on the server side, passing the distance matrix directly to the visualization process instead of the raw vectors.
|
||||
|
||||
While waiting for the distance-matrix API to be ready, I explored further optimizations. I observed that the worker thread sent results to the main thread for rendering at specific intervals, causing unnecessary delays due to serialization and deserialization.
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/channels.png"
|
||||
caption="Serialization and Deserialization Overhead"
|
||||
alt="Image showing serialization and deserialization overhead due to message passing between threads"
|
||||
>}}
|
||||
|
||||
To address this, I implemented a `SharedArrayBuffer`, allowing the main thread to access changes made by the worker thread instantly. This change led to noticeable improvements.
|
||||
|
||||
Additionally, the previous architecture resulted in choppy animations due to the fixed intervals at which the worker thread sent results.
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/prev_arch.png"
|
||||
caption="Previous architecture with fixed intervals"
|
||||
alt="Image showing the previous architecture of the frontend with fixed intervals for sending results"
|
||||
>}}
|
||||
|
||||
I introduced a "rendering-on-demand" approach, where the main thread would signal the worker thread when it was ready to render the next result. This created smoother, more responsive animations.
|
||||
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/curr_arch.png"
|
||||
caption="Current architecture with rendering-on-demand"
|
||||
alt="Image showing the current architecture of the frontend with rendering-on-demand approach"
|
||||
>}}
|
||||
|
||||
With these optimizations in place, the final step was wrapping up the project by creating a Node.js [package](https://www.npmjs.com/package/wasm-dist-bhtsne). This package exposed the necessary interfaces to accept the distance matrix, perform calculations, and return the results, making the solution easy to integrate into various projects.
|
||||
|
||||
|
||||
## Areas for Improvement
|
||||
|
||||
While reflecting on this transformative journey, there are still areas that offer room for improvement and future enhancements:
|
||||
|
||||
1. **Payload Parsing:** When requesting a large number of vectors, parsing the payload on the main thread can make the user interface unresponsive. Implementing a faster parser could mitigate this issue.
|
||||
|
||||
2. **Direct Data Requests:** Allowing the worker thread to request data directly could eliminate the initial transfer of data from the main thread, speeding up the overall process.
|
||||
|
||||
3. **Chart Library Optimization:** Profiling revealed that nearly 80% of the time was spent on the Chart.js update function. Switching to a WebGL-accelerated chart library could dramatically improve performance, especially for large datasets.
|
||||
{{< figure
|
||||
src="/articles_data/dimension-reduction-qsoc/profiling.png"
|
||||
caption="Profiling Result"
|
||||
alt="Image showing profiling results with 80% time spent on Chart.js update function"
|
||||
>}}
|
||||
|
||||
|
||||
## Conclusion
|
||||
|
||||
Participating in the Qdrant Summer of Code 2024 was a deeply rewarding experience. I had the chance to push the boundaries of my coding skills while exploring new technologies like Rust and WebAssembly. I'm incredibly grateful for the guidance and support from my mentor and the entire Qdrant team, who made this journey both educational and enjoyable.
|
||||
|
||||
This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide.
|
||||
|
||||
This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I've gained to future projects and to see how Qdrant's enhanced vector visualization feature will benefit users worldwide.
|
||||
|
||||
Thank you for joining me on this coding adventure. I hope you found something valuable in my journey, and I look forward to sharing more exciting projects with you in the future. Happy coding!
|
||||
@@ -51,7 +51,7 @@ Things have changed since then, as so many of you wanted a single tool for spars
|
||||
|
||||
If you're coming across the topic of sparse vectors for the first time, our [Brief History of Search](/documentation/overview/vector-search/) explains the difference between sparse and dense vectors.
|
||||
|
||||
Check out the [sparse vectors article](../sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users.
|
||||
Check out the [sparse vectors article](/articles/sparse-vectors/) and [sparse vectors index docs](/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users.
|
||||
|
||||
### Discovery API
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ keywords:
|
||||
|
||||
In today's fast-paced, information-rich world, AI is revolutionizing knowledge management. The systematic process of capturing, distributing, and effectively using knowledge within an organization is one of the fields in which AI provides exceptional value today.
|
||||
|
||||
> The potential for AI-powered knowledge management increases when leveraging Retrieval Augmented Generation (RAG), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases.
|
||||
> The potential for AI-powered knowledge management increases when leveraging [Retrieval Augmented Generation (RAG)](https://qdrant.tech/rag/rag-evaluation-guide/), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases.
|
||||
|
||||
This process enhances the accuracy, relevance, and reliability of generated text, thereby mitigating the risk of faulty, incorrect, or nonsensical results sometimes associated with traditional LLMs. This method not only ensures that the answers are contextually relevant but also up-to-date, reflecting the latest insights and data available.
|
||||
|
||||
@@ -35,7 +35,7 @@ In this article, we’ll break down a RAG Optimization workflow experiment that
|
||||
|
||||
Alongside Qdrant we will use Quotient, which provides a seamless way to evaluate your RAG implementation, accelerating and improving the experimentation process.
|
||||
|
||||
[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build evaluation frameworks and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users.
|
||||
[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build [evaluation frameworks](https://qdrant.tech/rag/rag-evaluation-guide/) and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users.
|
||||
|
||||
> 💡 The [accompanying notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient) for this exercise can be found on GitHub for future reference.
|
||||
|
||||
@@ -54,7 +54,7 @@ To evaluate a RAG pipeline , we will have to build a RAG Pipeline first. In the
|
||||
|
||||

|
||||
|
||||
The illustration below depicts how we can leverage a RAG Evaluation framework to assess the quality of RAG Application.
|
||||
The illustration below depicts how we can leverage a [RAG Evaluation framework](https://qdrant.tech/rag/rag-evaluation-guide/) to assess the quality of RAG Application.
|
||||
|
||||

|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@ tags:
|
||||
|
||||
**Semantic cache** is a method of retrieval optimization, where similar queries instantly retrieve the same appropriate response from a knowledge base.
|
||||
|
||||
Semantic cache differs from traditional caching methods. In computing, **cache** refers to high-speed memory that efficiently stores frequently accessed data. In the context of vector databases, a **semantic cache** improves AI application performance by storing previously retrieved results along with the conditions under which they were computed. This allows the application to reuse those results when the same or similar conditions occur again, rather than finding them from scratch.
|
||||
Semantic cache differs from traditional caching methods. In computing, **cache** refers to high-speed memory that efficiently stores frequently accessed data. In the context of [vector databases](/articles/what-is-a-vector-database/), a **semantic cache** improves AI application performance by storing previously retrieved results along with the conditions under which they were computed. This allows the application to reuse those results when the same or similar conditions occur again, rather than finding them from scratch.
|
||||
|
||||
> The term **"semantic"** implies that the cache takes into account the meaning or semantics of the data or computation being cached, rather than just its syntactic representation. This can lead to more efficient caching strategies that exploit the structure or relationships within the data or computation.
|
||||
|
||||
@@ -40,9 +40,9 @@ In this blog and video, we will walk you through how to use Qdrant to implement
|
||||
|
||||
Semantic cache is increasingly used in Retrieval-Augmented Generation (RAG) applications. In RAG, when a user asks a question, we embed it and search our vector database, either by using keyword, semantic, or hybrid search methods. The matched context is then passed to a Language Model (LLM) along with the prompt and user question for response generation.
|
||||
|
||||
Qdrant is recommended for setting up semantic cache as semantically evaluates the response. When semantic cache is implemented, we store common questions and their corresponding answers in a key-value cache. This way, when a user asks a question, we can retrieve the response from the cache if it already exists.
|
||||
Qdrant is recommended for setting up semantic cache as semantically [evaluates](https://qdrant.tech/rag/rag-evaluation-guide/) the response. When semantic cache is implemented, we store common questions and their corresponding answers in a key-value cache. This way, when a user asks a question, we can retrieve the response from the cache if it already exists.
|
||||
|
||||
**Diagram:** Semantic cache improves RAG by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers.
|
||||
**Diagram:** Semantic cache improves [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) by directly retrieving stored answers to the user. **Follow along with the gif** and see how semantic cache stores and retrieves answers.
|
||||
|
||||

|
||||
|
||||
|
||||
@@ -0,0 +1,531 @@
|
||||
---
|
||||
title: "What is Vector Quantization?"
|
||||
draft: false
|
||||
slug: what-is-vector-quantization
|
||||
short_description: What is Vector Quantization? Methods & Examples
|
||||
description: In this article, we'll teach you about compression methods like Scalar, Product, and Binary Quantization. Learn how to choose the best method for your specific application.
|
||||
preview_dir: /articles_data/what-is-vector-quantization/preview
|
||||
weight: -210
|
||||
social_preview_image: /articles_data/what-is-vector-quantization/preview/social-preview.jpg
|
||||
date: 2024-09-25T09:29:33-03:00
|
||||
author: Sabrina Aquino
|
||||
featured: true
|
||||
tags:
|
||||
- vector-search
|
||||
- vector-quantization
|
||||
- binary quantization
|
||||
- product quantization
|
||||
- scalar quantization
|
||||
- vector compression
|
||||
|
||||
---
|
||||
|
||||
Vector quantization is a data compression technique used to reduce the size of high-dimensional data. Compressing vectors reduces memory usage while maintaining nearly all of the essential information. This method allows for more efficient storage and faster search operations, particularly in large datasets.
|
||||
|
||||
When working with high-dimensional vectors, such as embeddings from providers like OpenAI, a single 1536-dimensional vector requires **6 KB of memory**.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/vector-size.png" alt="1536-dimensional vector size is 6 KB" width="700">
|
||||
|
||||
With 1 million vectors needing around 6 GB of memory, as your dataset grows to multiple **millions of vectors**, the memory and processing demands increase significantly.
|
||||
|
||||
To understand why this process is so computationally demanding, let's take a look at the nature of the [HNSW index](https://qdrant.tech/documentation/concepts/indexing/#vector-index).
|
||||
|
||||
The **HNSW (Hierarchical Navigable Small World) index** organizes vectors in a layered graph, connecting each vector to its nearest neighbors. At each layer, the algorithm narrows down the search area until it reaches the lower layers, where it efficiently finds the closest matches to the query.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/hnsw.png" alt="HNSW Search visualization" width="500">
|
||||
|
||||
Each time a new vector is added, the system must determine its position in the existing graph, a process similar to searching. This makes both inserting and searching for vectors complex operations.
|
||||
|
||||
One of the key challenges with the HNSW index is that it requires a lot of **random reads** and **sequential traversals** through the graph. This makes the process computationally expensive, especially when you're dealing with millions of high-dimensional vectors.
|
||||
|
||||
The system has to jump between various points in the graph in an unpredictable manner. This unpredictability makes optimization difficult, and as the dataset grows, the memory and processing requirements increase significantly.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/hnsw-search2.png" alt="HNSW Search visualization" width="600">
|
||||
|
||||
Since vectors need to be stored in **fast storage** like **RAM** or **SSD** for low-latency searches, as the size of the data grows, so does the cost of storing and processing it efficiently.
|
||||
|
||||
**Quantization** offers a solution by compressing vectors to smaller memory sizes, making the process more efficient.
|
||||
|
||||
There are several methods to achieve this, and here we will focus on three main ones:
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/types-of-quant.png" alt="Types of Quantization: 1. Scalar Quantization, 2. Product Quantization, 3. Binary Quantization" width="700">
|
||||
|
||||
## 1. What is Scalar Quantization?
|
||||
|
||||

|
||||
|
||||
In Qdrant, each dimension is represented by a `float32` value, which uses **4 bytes** of memory. When using [Scalar Quantization](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization), we map our vectors to a range that the smaller `int8` type can represent. An `int8` is only **1 byte** and can represent 256 values (from -128 to 127, or 0 to 255). This results in a **75% reduction** in memory size.
|
||||
|
||||
For example, if our data lies in the range of -1.0 to 1.0, Scalar Quantization will transform these values to a range that `int8` can represent, i.e., within -128 to 127. The system **maps** the `float32` values into this range.
|
||||
|
||||
Here's a simple linear example of what this process looks like:
|
||||
|
||||

|
||||
|
||||
To set up Scalar Quantization in Qdrant, you need to include the `quantization_config` section when creating or updating a collection:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 128,
|
||||
"distance": "Cosine"
|
||||
},
|
||||
"quantization_config": {
|
||||
"scalar": {
|
||||
"type": "int8",
|
||||
"quantile": 0.99,
|
||||
"always_ram": true
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
vectors_config=models.VectorParams(size=128, distance=models.Distance.COSINE),
|
||||
quantization_config=models.ScalarQuantization(
|
||||
scalar=models.ScalarQuantizationConfig(
|
||||
type=models.ScalarType.INT8,
|
||||
quantile=0.99,
|
||||
always_ram=True,
|
||||
),
|
||||
),
|
||||
)
|
||||
```
|
||||
|
||||
The `quantile` parameter is used to calculate the quantization bounds. For example, if you specify a `0.99` quantile, the most extreme 1% of values will be excluded from the quantization bounds.
|
||||
|
||||
This parameter only affects the resulting precision, not the memory footprint. You can adjust it if you experience a significant decrease in search quality.
|
||||
|
||||
Scalar Quantization is a great choice if you're looking to boost search speed and compression without losing much accuracy. It also slightly improves performance, as distance calculations (such as dot product or cosine similarity) using `int8` values are computationally simpler than using `float32` values.
|
||||
|
||||
While the performance gains of Scalar Quantization may not match those achieved with Binary Quantization (which we'll discuss later), it remains an excellent default choice when Binary Quantization isn’t suitable for your use case.
|
||||
|
||||
## 2. What is Binary Quantization?
|
||||
|
||||

|
||||
|
||||
[Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) is an excellent option if you're looking to **reduce memory** usage while also achieving a significant **boost in speed**. It works by converting high-dimensional vectors into simple binary (0 or 1) representations.
|
||||
|
||||
- Values greater than zero are converted to 1.
|
||||
- Values less than or equal to zero are converted to 0.
|
||||
|
||||
Let's consider our initial example of a 1536-dimensional vector that requires **6 KB** of memory (4 bytes for each `float32` value).
|
||||
|
||||
After Binary Quantization, each dimension is reduced to 1 bit (1/8 byte), so the memory required is:
|
||||
|
||||
$$
|
||||
\frac{1536 \text{ dimensions}}{8 \text{ bits per byte}} = 192 \text{ bytes}
|
||||
$$
|
||||
|
||||
This leads to a **32x** memory reduction.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/binary-quant.png" alt="Binary Quantization example" width="800">
|
||||
|
||||
Qdrant automates the Binary Quantization process during indexing. As vectors are added to your collection, each 32-bit floating-point component is converted into a binary value according to the configuration you define.
|
||||
|
||||
Here’s how you can set it up:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1536,
|
||||
"distance": "Cosine"
|
||||
},
|
||||
"quantization_config": {
|
||||
"binary": {
|
||||
"always_ram": true
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE),
|
||||
quantization_config=models.BinaryQuantization(
|
||||
binary=models.BinaryQuantizationConfig(
|
||||
always_ram=True,
|
||||
),
|
||||
),
|
||||
)
|
||||
```
|
||||
|
||||
Binary Quantization is by far the quantization method that provides the most significant processing **speed gains** compared to Scalar and Product Quantizations. This is because the binary representation allows the system to use highly optimized CPU instructions, such as [XOR](https://en.wikipedia.org/wiki/XOR_gate#:~:text=XOR%20represents%20the%20inequality%20function,the%20other%20but%20not%20both%22) and [Popcount](https://en.wikipedia.org/wiki/Hamming_weight), for fast distance computations.
|
||||
|
||||
It can speed up search operations by **up to 40x**, depending on the dataset and hardware.
|
||||
|
||||
Not all models are equally compatible with Binary Quantization, and in the comparison above, we are only using models that are compatible. Some models may experience a greater loss in accuracy when quantized. We recommend using Binary Quantization with models that have **at least 1024 dimensions** to minimize accuracy loss.
|
||||
|
||||
The models that have shown the best compatibility with this method include:
|
||||
|
||||
- **OpenAI text-embedding-ada-002** (1536 dimensions)
|
||||
- **Cohere AI embed-english-v2.0** (4096 dimensions)
|
||||
|
||||
These models demonstrate minimal accuracy loss while still benefiting from substantial speed and memory gains.
|
||||
|
||||
Even though Binary Quantization is incredibly fast and memory-efficient, the trade-offs are in **precision** and **model compatibility**, so you may need to ensure search quality using techniques like oversampling and rescoring.
|
||||
|
||||
If you're interested in exploring Binary Quantization in more detail—including implementation examples, benchmark results, and usage recommendations—check out our dedicated article on [Binary Quantization - Vector Search, 40x Faster](https://qdrant.tech/articles/binary-quantization/).
|
||||
|
||||
## 3. What is Product Quantization?
|
||||
|
||||

|
||||
|
||||
[Product Quantization](https://qdrant.tech/documentation/guides/quantization/#product-quantization) is a method used to compress high-dimensional vectors by representing them with a smaller set of representative points.
|
||||
|
||||
The process begins by splitting the original high-dimensional vectors into smaller **sub-vectors.** Each sub-vector represents a segment of the original vector, capturing different characteristics of the data.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/subvec.png" alt="Creation of the Sub-vector" width="700">
|
||||
|
||||
For each sub-vector, a separate **codebook** is created, representing regions in the data space where common patterns occur.
|
||||
|
||||
The codebook in Qdrant is trained automatically during the indexing process. As vectors are added to the collection, Qdrant uses your specified quantization settings in the `quantization_config` to build the codebook and quantize the vectors. Here’s how you can set it up:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1024,
|
||||
"distance": "Cosine"
|
||||
},
|
||||
"quantization_config": {
|
||||
"product": {
|
||||
"compression": "x32",
|
||||
"always_ram": true
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
```python
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
vectors_config=models.VectorParams(size=1024, distance=models.Distance.COSINE),
|
||||
quantization_config=models.ProductQuantization(
|
||||
product=models.ProductQuantizationConfig(
|
||||
compression=models.CompressionRatio.X32,
|
||||
always_ram=True,
|
||||
),
|
||||
),
|
||||
)
|
||||
```
|
||||
|
||||
Each region in the codebook is defined by a **centroid**, which serves as a representative point summarizing the characteristics of that region. Instead of treating every single data point as equally important, we can group similar sub-vectors together and represent them with a single centroid that captures the general characteristics of that group.
|
||||
|
||||
The centroids used in Product Quantization are determined using the **[K-means clustering algorithm](https://en.wikipedia.org/wiki/K-means_clustering)**.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/code-book.png" alt="Codebook and Centroids example" width="700">
|
||||
|
||||
Qdrant always selects **K = 256** as the number of centroids in its implementation, based on the fact that 256 is the maximum number of unique values that can be represented by a single byte.
|
||||
|
||||
This makes the compression process efficient because each centroid index can be stored in a single byte.
|
||||
|
||||
The original high-dimensional vectors are quantized by mapping each sub-vector to the nearest centroid in its respective codebook.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/mapping.png" alt="Vectors being mapped to their corresponding centroids example" width="700">
|
||||
|
||||
The compressed vector stores the index of the closest centroid for each sub-vector.
|
||||
|
||||
Here’s how a 1024-dimensional vector, originally taking up 4096 bytes, is reduced to just 128 bytes by representing it as 128 indexes, each pointing to the centroid of a sub-vector:
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/product-quant.png" alt="Product Quantization example" width="800">
|
||||
|
||||
After setting up quantization and adding your vectors, you can perform searches as usual. Qdrant will automatically use the quantized vectors, optimizing both speed and memory usage. Optionally, you can enable rescoring for better accuracy.
|
||||
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/search
|
||||
{
|
||||
"query": [0.22, -0.01, -0.98, 0.37],
|
||||
"params": {
|
||||
"quantization": {
|
||||
"rescore": true
|
||||
}
|
||||
},
|
||||
"limit": 10
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.query_points(
|
||||
collection_name="my_collection",
|
||||
query_vector=[0.22, -0.01, -0.98, 0.37], # Your query vector
|
||||
search_params=models.SearchParams(
|
||||
quantization=models.QuantizationSearchParams(
|
||||
rescore=True # Enables rescoring with original vectors
|
||||
)
|
||||
),
|
||||
limit=10 # Return the top 10 results
|
||||
)
|
||||
```
|
||||
Product Quantization can significantly reduce memory usage, potentially offering up to **64x** compression in certain configurations. However, it's important to note that this level of compression can lead to a noticeable drop in quality.
|
||||
|
||||
If your application requires high precision or real-time performance, Product Quantization may not be the best choice. However, if **memory savings** are critical and some accuracy loss is acceptable, it could still be an ideal solution.
|
||||
|
||||
Here’s a comparison of speed, accuracy, and compression for all three methods, adapted from [Qdrant's documentation](https://qdrant.tech/documentation/guides/quantization/#how-to-choose-the-right-quantization-method):
|
||||
|
||||
| Quantization method | Accuracy | Speed | Compression |
|
||||
|---------------------|----------|------------|-------------|
|
||||
| Scalar | 0.99 | up to x2 | 4 |
|
||||
| Product | 0.7 | 0.5 | up to 64 |
|
||||
| Binary | 0.95* | up to x40 | 32 |
|
||||
|
||||
\* - for compatible models
|
||||
|
||||
For a more in-depth understanding of the benchmarks you can expect, check out our dedicated article on [Product Quantization in Vector Search](https://qdrant.tech/articles/product-quantization/).
|
||||
|
||||
## Rescoring, Oversampling, and Reranking
|
||||
|
||||
When we use quantization methods like Scalar, Binary, or Product Quantization, we're compressing our vectors to save memory and improve performance. However, this compression removes some detail from the original vectors.
|
||||
|
||||
This can slightly reduce the accuracy of our similarity searches because the quantized vectors are approximations of the original data. To mitigate this loss of accuracy, you can use **oversampling** and **rescoring**, which help improve the accuracy of the final search results.
|
||||
|
||||
The original vectors are never deleted during this process, and you can easily switch between quantization methods or parameters by updating the collection configuration at any time.
|
||||
|
||||
Here’s how the process works, step by step:
|
||||
|
||||
### 1. Initial Quantized Search
|
||||
|
||||
When you perform a search, Qdrant retrieves the top candidates using the quantized vectors based on their similarity to the query vector, as determined by the quantized data. This step is fast because we're using the quantized vectors.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/ann-search-quantized.png" alt="ANN Search with Quantization" width="600">
|
||||
|
||||
### 2. Oversampling
|
||||
|
||||
Oversampling is a technique that helps compensate for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results.
|
||||
|
||||
You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results (`limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2).
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/ann-search-quantized-oversampling.png" alt="ANN Search with Quantization and Oversampling" width="600">
|
||||
|
||||
You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of obtaining high-quality top-K results, especially after rescoring with the original vectors.
|
||||
|
||||
### 3. Rescoring with Original Vectors
|
||||
|
||||
After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query.
|
||||
|
||||
The rescoring process **maps** the quantized vectors to their corresponding original vectors, allowing you to consider factors like context, metadata, or additional relevance that wasn’t included in the initial search, leading to more accurate results.
|
||||
|
||||

|
||||
|
||||
During rescoring, one of the lower-ranked candidates from oversampling might turn out to be a better match than some of the original top-K candidates.
|
||||
|
||||
Even though rescoring uses the original, larger vectors, the process remains much faster because only a very small number of vectors are read. The initial quantized search already identifies the specific vectors to read, rescore, and rerank.
|
||||
|
||||
### 4. Reranking
|
||||
|
||||
With the new similarity scores from rescoring, **reranking** is where the final top-K candidates are determined based on the updated similarity scores.
|
||||
|
||||
For example, in our case with a limit of 4, a candidate that ranked 6th in the initial quantized search might improve its score after rescoring because the original vectors capture more context or metadata. As a result, this candidate could move into the final top 4 after reranking, replacing a less relevant option from the initial search.
|
||||
|
||||
<img src="/articles_data/what-is-vector-quantization/reranking.png" alt="Reranking with Original Vectors" width="600">
|
||||
|
||||
Here's how you can set it up:
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/search
|
||||
|
||||
|
||||
{
|
||||
"query": [0.22, -0.01, -0.98, 0.37],
|
||||
"params": {
|
||||
"quantization": {
|
||||
"rescore": true,
|
||||
"oversampling": 2
|
||||
}
|
||||
},
|
||||
"limit": 4
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.query_points(
|
||||
collection_name="my_collection",
|
||||
query_vector=[0.22, -0.01, -0.98, 0.37],
|
||||
search_params=models.SearchParams(
|
||||
quantization=models.QuantizationSearchParams(
|
||||
rescore=True, # Enables rescoring with original vectors
|
||||
oversampling=2 # Retrieves extra candidates for rescoring
|
||||
)
|
||||
),
|
||||
limit=4 # Desired number of final results
|
||||
)
|
||||
```
|
||||
|
||||
You can adjust the `oversampling` factor to find the right balance between search speed and result accuracy.
|
||||
|
||||
If quantization is impacting performance in an application that requires high accuracy, combining oversampling with rescoring is a great choice. However, if you need faster searches and can tolerate some loss in accuracy, you might choose to use oversampling without rescoring, or adjust the oversampling factor to a lower value.
|
||||
|
||||
## Distributing Resources Between Disk & Memory
|
||||
|
||||
Qdrant stores both the quantized and original vectors. When you enable quantization, both the original and quantized vectors are stored in RAM by default. You can move the original vectors to disk to significantly reduce RAM usage and lower system costs. Simply enabling quantization is not enough—you need to explicitly move the original vectors to disk by setting `on_disk=True`.
|
||||
|
||||
Here’s an example configuration:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1536,
|
||||
"distance": "Cosine",
|
||||
"on_disk": true # Move original vectors to disk
|
||||
},
|
||||
"quantization_config": {
|
||||
"binary": {
|
||||
"always_ram": true # Store only quantized vectors in RAM
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.update_collection(
|
||||
collection_name="my_collection",
|
||||
vectors_config=models.VectorParams(
|
||||
size=1536,
|
||||
distance=models.Distance.COSINE,
|
||||
on_disk=True # Move original vectors to disk
|
||||
),
|
||||
quantization_config=models.BinaryQuantization(
|
||||
binary=models.BinaryQuantizationConfig(
|
||||
always_ram=True # Store only quantized vectors in RAM
|
||||
)
|
||||
)
|
||||
)
|
||||
```
|
||||
|
||||
Without explicitly setting `on_disk=True`, you won't see any RAM savings, even with quantization enabled. So, make sure to configure both storage and quantization options based on your memory and performance needs. If your storage has high disk latency, consider disabling rescoring to maintain speed.
|
||||
|
||||
### Speeding Up Rescoring with io_uring
|
||||
|
||||
When dealing with large collections of quantized vectors, frequent disk reads are required to retrieve both original and compressed data for rescoring operations. While `mmap` helps with efficient I/O by reducing user-to-kernel transitions, rescoring can still be slowed down when working with large datasets on disk due to the need for frequent disk reads.
|
||||
|
||||
On Linux-based systems, `io_uring` allows multiple disk operations to be processed in parallel, significantly reducing I/O overhead. This optimization is particularly effective during rescoring, where multiple vectors need to be re-evaluated after the initial search. With io_uring, Qdrant can retrieve and rescore vectors from disk in the most efficient way, improving overall search performance.
|
||||
|
||||
When you perform vector quantization and store data on disk, Qdrant often needs to access multiple vectors in parallel. Without io_uring, this process can be slowed down due to the system’s limitations in handling many disk accesses.
|
||||
|
||||
To enable `io_uring` in Qdrant, add the following to your storage configuration:
|
||||
|
||||
```yaml
|
||||
storage:
|
||||
async_scorer: true # Enable io_uring for async storage
|
||||
```
|
||||
|
||||
Without this configuration, Qdrant will default to using `mmap` for disk I/O operations.
|
||||
|
||||
For more information and benchmarks comparing io_uring with traditional I/O approaches like mmap, check out [Qdrant's io_uring implementation article.](https://qdrant.tech/articles/io_uring/)
|
||||
|
||||
## Performance of Quantized vs. Non-Quantized Data
|
||||
|
||||
Qdrant uses the quantized vectors by default if they are available. If you want to evaluate how quantization affects your search results, you can temporarily disable it to compare results from quantized and non-quantized searches. To do this, set `ignore: true` in the query:
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
{
|
||||
"query": [0.22, -0.01, -0.98, 0.37],
|
||||
"params": {
|
||||
"quantization": {
|
||||
"ignore": true,
|
||||
}
|
||||
},
|
||||
"limit": 4
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.query_points(
|
||||
collection_name="{collection_name}",
|
||||
query=[0.22, -0.01, -0.98, 0.37],
|
||||
search_params=models.SearchParams(
|
||||
quantization=models.QuantizationSearchParams(
|
||||
ignore=True
|
||||
)
|
||||
),
|
||||
)
|
||||
```
|
||||
### Switching Between Quantization Methods
|
||||
|
||||
Not sure if you’ve chosen the right quantization method? In Qdrant, you have the flexibility to remove quantization and rely solely on the original vectors, adjust the quantization type, or change compression parameters at any time without affecting your original vectors.
|
||||
|
||||
To switch to binary quantization and adjust the compression rate, for example, you can update the collection’s quantization configuration using the `update_collection` method:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1536,
|
||||
"distance": "Cosine"
|
||||
},
|
||||
"quantization_config": {
|
||||
"binary": {
|
||||
"always_ram": true,
|
||||
"compression_rate": 0.8 # Set the new compression rate
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
```python
|
||||
client.update_collection(
|
||||
collection_name="my_collection",
|
||||
quantization_config=models.BinaryQuantization(
|
||||
binary=models.BinaryQuantizationConfig(
|
||||
always_ram=True, # Store only quantized vectors in RAM
|
||||
compression_rate=0.8 # Set the new compression rate
|
||||
)
|
||||
),
|
||||
)
|
||||
```
|
||||
|
||||
If you decide to **turn off quantization** and use only the original vectors, you can remove the quantization settings entirely with `quantization_config=None`:
|
||||
|
||||
```http
|
||||
PUT /collections/my_collection
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1536,
|
||||
"distance": "Cosine"
|
||||
},
|
||||
"quantization_config": null # Remove quantization and use original vectors only
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
client.update_collection(
|
||||
collection_name="my_collection",
|
||||
quantization_config=None # Remove quantization and rely on original vectors only
|
||||
)
|
||||
```
|
||||
## Wrapping Up
|
||||
|
||||

|
||||
|
||||
Quantization methods like Scalar, Product, and Binary Quantization offer powerful ways to optimize memory usage and improve search performance when dealing with large datasets of high-dimensional vectors. Each method comes with its own trade-offs between memory savings, computational speed, and accuracy.
|
||||
|
||||
Here are some final thoughts to help you choose the right quantization method for your needs:
|
||||
|
||||
| **Quantization Method** | **Key Features** | **When to Use** |
|
||||
|--------------------------|-------------------------------------------------------------|--------------------------------------------------------------------------------------------|
|
||||
| **Binary Quantization** | • **Fastest method and most memory-efficient**<br>• Up to **40x** faster search and **32x** reduced memory footprint | • Use with tested models like OpenAI's `text-embedding-ada-002` and Cohere's `embed-english-v2.0`<br>• When speed and memory efficiency are critical |
|
||||
| **Scalar Quantization** | • **Minimal loss of accuracy**<br>• Up to **4x** reduced memory footprint | • Safe default choice for most applications.<br>• Offers a good balance between accuracy, speed, and compression. |
|
||||
| **Product Quantization** | • **Highest compression ratio**<br>• Up to **64x** reduced memory footprint | • When minimizing memory usage is the top priority<br>• Acceptable if some loss of accuracy and slower indexing is tolerable |
|
||||
|
||||
### Learn More
|
||||
|
||||
If you want to learn more about improving accuracy, memory efficiency, and speed when using quantization in Qdrant, we have a dedicated [Quantization tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section in our docs that explains all the quantization tips you can use to enhance your results.
|
||||
|
||||
Learn more about optimizing real-time precision with oversampling in Binary Quantization by watching this interview with Qdrant’s CTO, Andrey Vasnetsov:
|
||||
|
||||
<div style="position: relative; padding-bottom: 56.25%; height: 0; overflow: hidden;">
|
||||
<iframe src="https://www.youtube.com/embed/4aUq5VnR_VI" frameborder="0" allowfullscreen style="position: absolute; top: 0; left: 0; width: 100%; height: 90%;">
|
||||
</iframe>
|
||||
</div>
|
||||
|
||||
Stay up-to-date on the latest in [vector search](/advanced-search/) and quantization, share your projects, ask questions, [join our vector search community](https://discord.com/invite/qdrant)!
|
||||
@@ -34,7 +34,7 @@ While you could be more creative with your prompts, it is only a short-term solu
|
||||
|
||||
The image above shows how a basic RAG system works. Before forwarding the question to the LLM, we have a layer that searches our knowledge base for the "relevant knowledge" to answer the user query. Specifically, in this case, the spending data from the last month. Our LLM can now generate a **relevant non-hallucinated** response about our budget.
|
||||
|
||||
As your data grows, you’ll need efficient ways to identify the most relevant information for your LLM's limited memory. This is where you’ll want a proper way to store and retrieve the specific data you’ll need for your query, without needing the LLM to remember it.
|
||||
As your data grows, you’ll need [efficient ways](https://qdrant.tech/rag/rag-evaluation-guide/) to identify the most relevant information for your LLM's limited memory. This is where you’ll want a proper way to store and retrieve the specific data you’ll need for your query, without needing the LLM to remember it.
|
||||
|
||||
**Vector databases** store information as **vector embeddings**. This format supports efficient similarity searches to retrieve relevant data for your query. For example, Qdrant is specifically designed to perform fast, even in scenarios dealing with billions of vectors.
|
||||
|
||||
@@ -167,7 +167,7 @@ Are you ready to create your own RAG chatbot from the ground up? We have a video
|
||||
* Applying vector similarity search algorithms
|
||||
* Enhancing the efficiency and response quality
|
||||
|
||||
After building your RAG chatbot, you'll be able to evaluate its performance against that of a chatbot powered solely by a Large Language Model (LLM).
|
||||
After building your RAG chatbot, you'll be able to [evaluate its performance](https://qdrant.tech/rag/rag-evaluation-guide/) against that of a chatbot powered solely by a Large Language Model (LLM).
|
||||
|
||||
<div style="max-width: 640px; margin: 0 auto; padding-bottom: 1em"> <div style="position: relative; padding-bottom: 56.25%; height: 0; overflow: hidden;"> <iframe width="100%" height="100%" src="https://www.youtube.com/embed/O60-KuZZeQA" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" allowfullscreen style="position: absolute; top: 0; left: 0; width: 100%; height: 100%;"></iframe> </div> </div>
|
||||
|
||||
|
||||
@@ -50,7 +50,7 @@ We're incredibly excited about this collaboration with Azure Marketplace and the
|
||||
|
||||
Ready to elevate your business with Qdrant? **Click the banner and get started today!**
|
||||
|
||||
[](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db)
|
||||
[](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db)
|
||||
|
||||
### About Qdrant:
|
||||
|
||||
|
||||
@@ -52,7 +52,8 @@ data usually sits in various SaaS applications across the organization.
|
||||
|
||||
Dust provides companies with the core platform to execute on their GenAI bet
|
||||
for their teams by deploying LLMs across the organization and providing context
|
||||
aware AI assistants through RAG. Users can manage so-called data sources within
|
||||
aware AI assistants through [RAG](https://qdrant.tech/rag/rag-evaluation-guide/)
|
||||
. Users can manage so-called data sources within
|
||||
Dust and upload files or directly connect to it via APIs to ingest data from
|
||||
tools like Notion, Google Drive, or Slack. Dust then handles the chunking
|
||||
strategy with the embeddings models and performs retrieval augmented generation.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Kern AI & Qdrant: Precision AI Solutions for Finance and Insurance"
|
||||
short_description: "Transforming customer service in finance and insurance with vector search-based retrieval.</p>"
|
||||
short_description: "Transforming customer service in finance and insurance with vector search-based retrieval."
|
||||
description: "Revolutionizing customer service in finance and insurance by leveraging vector search for faster responses and improved operational efficiency."
|
||||
preview_image: /blog/case-study-kern/preview.png
|
||||
social_preview_image: /blog/case-study-kern/preview.png
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Nyris & Qdrant: How Vectors are the Future of Visual Search"
|
||||
short_description: "Transforming customer service in finance and insurance with vector search-based retrieval.</p>"
|
||||
short_description: "Transforming customer service in finance and insurance with vector search-based retrieval."
|
||||
description: "Revolutionizing customer service in finance and insurance by leveraging vector search for faster responses and improved operational efficiency."
|
||||
preview_image: /blog/case-study-nyris/preview.png
|
||||
social_preview_image: /blog/case-study-nyris/preview.png
|
||||
@@ -34,7 +34,7 @@ During his time at Amazon, Lukasson observed that search engines like Google oft
|
||||
|
||||
In their quest for the perfect visual search provider, Nyris ultimately decided to develop their own solution.
|
||||
|
||||
## The Path to Vector-based Visual Search
|
||||
## The Path to Vector-Based Visual Search
|
||||
|
||||
Initially in 2015, the team explored traditional search algorithms based on key value SIFT (Scale Invariant Feature Transform) features to locate specific elements within images. However, they quickly realized that these methods were imprecise and unreliable. To address this, Nyris began experimenting with the first Convolutional Neural Networks (CNNs) to extract embeddings for vector search.
|
||||
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Qdrant and Shakudo: Secure & Performant Vector Search in VPC Environments"
|
||||
short_description: "Transforming customer service in finance and insurance with vector search-based retrieval."
|
||||
description: "Implementing vector search for enterprise AI via Qdrant's Hybrid Cloud integration into Shakudo’s virtual private cloud."
|
||||
preview_image: /blog/case-study-shakudo/preview.png
|
||||
social_preview_image: /blog/case-study-shakudo/preview.png
|
||||
date: 2024-09-23T00:02:00Z
|
||||
author: Qdrant
|
||||
featured: false
|
||||
tags:
|
||||
- Shakudo
|
||||
- Vector Search
|
||||
---
|
||||
|
||||
We are excited to announce that Qdrant has partnered with [Shakudo](https://www.shakudo.io/), bringing [Qdrant Hybrid Cloud](https://qdrant.tech/hybrid-cloud/) to Shakudo’s virtual private cloud (VPC) deployments. This collaboration allows Shakudo clients to seamlessly integrate Qdrant’s high-performance vector database as a managed service into their private infrastructure, ensuring data sovereignty, scalability, and low-latency vector search for enterprise AI applications.
|
||||
|
||||
## Data Sovereignty and Compliance with Secure Vector Search
|
||||
|
||||
Shakudo’s VPC deployments ensure that client data remains within their infrastructure, providing strict control over sensitive information while leveraging a fully managed AI toolset. Qdrant Hybrid Cloud is tailored for environments where data privacy and regulatory compliance are paramount. It keeps the data plane inside the customer's infrastructure, with only essential telemetry shared externally, guaranteeing database isolation and security, while providing a fully managed service.
|
||||
|
||||

|
||||
|
||||
## Scaling and Performance Optimization for Enterprise Vector Search
|
||||
|
||||
Qdrant Hybrid Cloud is optimized for Kubernetes, allowing for fast, automated deployments and hands-off cluster management. Shakudo’s platform, designed for VPC-based environments, allows businesses to deploy Qdrant’s vector search clusters with no DevOps overhead. Qdrant’s ability to handle billions of vectors - powered by our customized Hierarchical Navigable Small World (HNSW) indexing - ensures real-time processing and high accuracy for AI-driven applications like semantic search, recommendation systems, and retrieval-augmented generation (RAG).
|
||||
|
||||
## Staying Compatible with the Entire Stack
|
||||
|
||||
By deploying Qdrant Hybrid Cloud on Shakudo, organizations gain immediate compatibility with their existing data sources, pipelines, and applications. It integrates seamlessly with the existing stack, ensuring smooth and efficient operation across all components. As business needs evolve, the data stack can easily scale and adapt to new demands.
|
||||
|
||||
## Key Benefits of Qdrant in Shakudo's Virtual Private Cloud
|
||||
|
||||
- **Data Privacy & Control**: Shakudo users can run a Qdrant vector database inside their own VPC, ensuring sensitive data never leaves their infrastructure, while enjoying a managed service for simplicity and reliability.
|
||||
- **Seamless Integration**: Qdrant’s Kubernetes-native setup allows rapid deployment on Shakudo’s VPC-based infrastructure, which provides pre-configured environments optimized for AI workloads.
|
||||
- **Scalability**: Qdrant’s ability to handle billions of vectors and its high-performance indexing like HNSW make it ideal for applications requiring fast, accurate similarity searches.
|
||||
- **Enterprise Flexibility**: With both on-premise and cloud-native setups available, this partnership offers businesses the flexibility to balance operational needs with privacy requirements.
|
||||
|
||||
## Learn More
|
||||
|
||||
Ready to learn how Qdrant on Shakudo can enhance your AI infrastructure? Contact the Shakudo team to explore how they can help you deploy secure, high-performance vector search in your VPC environment, or get started [here](https://www.shakudo.io/integrations/qdrant).
|
||||
|
||||
If you are interested in Qdrant’s Managed Cloud, Hybrid Cloud, or Private Cloud solutions for flexible deployment options for top-tier data privacy, [contact us](https://qdrant.tech/contact-sales/).
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](/hybrid-cloud/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed vector database that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product.
|
||||
In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](/hybrid-cloud/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed [vector database](/articles/what-is-a-vector-database/) that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product.
|
||||
|
||||
This is a major step forward in offering enterprise customers incredible synergy for maximizing the potential of their AI data. Qdrant's new Kubernetes-native design, coupled with Airbyte’s powerful data ingestion pipelines meet the needs of developers who are both prototyping and building production-level apps. Airbyte simplifies the process of data integration by providing a platform that connects to various sources and destinations effortlessly. Moreover, Qdrant Hybrid Cloud leverages advanced indexing and search capabilities to empower users to explore and analyze their data efficiently.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering, allowing developers to deploy a managed vector database in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations.
|
||||
We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering, allowing developers to deploy a managed [vector database](/articles/what-is-a-vector-database/) in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations.
|
||||
|
||||
With a highly customizable framework like Haystack, implementing vector search becomes incredibly simple. Qdrant's new Qdrant Hybrid Cloud offering and its Kubernetes-native design supports customers all the way from a simple prototype setup to a production scenario on any hosting platform. Users can attach AI functionalities to their existing in-house software by creating custom integration components. Don’t forget, both products are open-source and highly modular!
|
||||
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
---
|
||||
draft: false
|
||||
title: "New DeepLearning.AI Course on Retrieval Optimization: From Tokenization to Vector Quantization"
|
||||
short_description: "Free, beginner-friendly course to learn retrieval optimization and boost search performance."
|
||||
description: "Join Qdrant and DeepLearning.AI’s free, beginner-friendly course to learn retrieval optimization and boost search performance in machine learning."
|
||||
preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg
|
||||
social_preview_image: /blog/qdrant-deeplearning-ai-course/preview.jpg
|
||||
date: 2024-10-06T00:02:00Z
|
||||
author: Qdrant
|
||||
featured: false
|
||||
tags:
|
||||
- DeepLearning.AI
|
||||
- Vector Search
|
||||
- Vector Quantization
|
||||
- Tokenization
|
||||
- Retrieval-Augmented Generation
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited to announce a new course on DeepLearning.AI's platform: [Retrieval Optimization: From Tokenization to Vector Quantization](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo). This collaboration between Qdrant and DeepLearning.AI aims to empower developers and data enthusiasts with the skills needed to enhance [vector search](/advanced-search/) capabilities in their applications.
|
||||
|
||||
Led by Qdrant’s Kacper Łukawski, this free, one-hour course is designed for beginners eager to delve into the world of retrieval optimization.
|
||||
|
||||
## Why This Collaboration Matters
|
||||
|
||||
At Qdrant, we believe in the power of effective search to transform user experiences. Partnering with DeepLearning.AI allows us to combine our cutting-edge vector search technology with their educational expertise, providing learners with a comprehensive understanding of how to build and optimize [Retrieval-Augmented Generation (RAG)](/rag/rag-evaluation-guide/) applications. This course is part of our commitment to equip the community with practical skills that leverage advanced machine learning techniques.
|
||||
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/AE8i69Kcodc?si=IdTEKlUHVbGzgJD-" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
|
||||
## What You’ll Learn
|
||||
|
||||
In this course, you’ll explore key concepts that will enhance your understanding of retrieval optimization:
|
||||
|
||||
- Learn how tokenization works in large language and embedding models and how the tokenizer can affect the quality of your search.
|
||||
- Explore how different tokenization techniques including Byte-Pair Encoding, WordPiece, and Unigram are trained and work.
|
||||
- Understand how to [measure the quality of your retrieval](/rag/rag-evaluation-guide/) and how to optimize your search by adjusting HNSW parameters and [vector quantizations](/articles/what-is-vector-quantization/).
|
||||
|
||||
## Who Should Enroll
|
||||
|
||||
This course is tailored for anyone with basic Python knowledge.
|
||||
|
||||
Whether you’re starting your journey in machine learning or looking to enhance your existing skills, this course offers valuable insights to boost your capabilities.
|
||||
|
||||
### At a Glance:
|
||||
|
||||
- **Speaker**: Kacper Łukawski, Qdrant Developer Advocate
|
||||
- **Level**: Beginner
|
||||
- **Cost**: Free
|
||||
- **Location**: Online
|
||||
- **Duration**: 1 Hour
|
||||
|
||||
## How to Enroll
|
||||
|
||||
[Enroll via the DeepLearning.AI website](https://www.deeplearning.ai/short-courses/retrieval-optimization-from-tokenization-to-vector-quantization/?utm_campaign=qdrant-launch&utm_medium=qdrant&utm_source=partner-promo).
|
||||
@@ -0,0 +1,84 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Introducing Qdrant for Startups"
|
||||
short_description: "Join our Startup Program now and scale your AI-driven applications with ease."
|
||||
description: "Enjoy special discounts from Qdrant, HuggingFace, LlamaIndex, and Airbyte, as well as expert support & tooling perks, and be the first to try new features."
|
||||
preview_image: /blog/qdrant-for-startups-launch/preview.png
|
||||
social_preview_image: /blog/qdrant-for-startups-launch/preview.png
|
||||
date: 2024-10-02T00:02:00Z
|
||||
author: Qdrant
|
||||
featured: false
|
||||
tags:
|
||||
- Qdrant
|
||||
- Startups
|
||||
- Vector Search
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
# Supporting Early-Stage Startups
|
||||
|
||||
Over the past few years, we’ve witnessed some of the most innovative AI applications being built on Qdrant. A significant number of these have come from startups pushing the boundaries of what’s possible in AI. To ensure these pioneering teams have access to the right resources at the right time, we're introducing **Qdrant for Startups**. This initiative is designed to provide startups with the technical support, guidance, and infrastructure they need to scale their AI innovations quickly and effectively.
|
||||
|
||||
Qdrant for Startups helps early-stage startups fully leverage the capabilities of vector search technology. Whether you're building retrieval-augmented generation (RAG) systems, recommendation engines, or anomaly detection models, the program offers exclusive benefits, such as discounts for Qdrant cloud, expert technical guidance, exclusive partner benefits, and co-marketing opportunities - empowering you to build and scale your AI products efficiently and cost-effectively.
|
||||
|
||||
## Benefits for admitted startups:
|
||||
|
||||
- **Qdrant Cloud discount:** 20% discount on Qdrant Cloud valid for 12 months, optimizing costs while scaling with advanced vector search capabilities.
|
||||
- **Expert technical guidance:** Dedicated technical support and guidance to optimize your application’s performance with vector search.
|
||||
- **Co-marketing opportunities:** Collaboration with the Qdrant team on joint marketing initiatives to boost your startup’s visibility.
|
||||
- **Early access to features:** Exclusive early access to upcoming Qdrant features, keeping you at the forefront of technological advancements.
|
||||
- **Community access:** Access to Qdrant’s developer and AI community for collaboration, networking, and shared learning.
|
||||
|
||||
## Access to popular AI tools
|
||||
|
||||
We’ve built this program to support startups with their entire AI tech stack. In addition to Qdrant, accepted startups will receive exclusive discounts from our program partners - Hugging Face, LlamaIndex, and Airbyte - ensuring you have access to the key tools and resources needed to build and scale AI-driven applications.
|
||||
|
||||
Accepted startup program members will have the ability to get additional benefits:
|
||||
|
||||
- Hugging Face: $100 compute credits for the HuggingFace Hub
|
||||
- LlamaIndex: 20% discount for 12 months for LlamaCloud
|
||||
- Airbyte: Cloud credits for Y Combinator startups
|
||||
|
||||
[](https://qdrant.tech/qdrant-for-startups)
|
||||
|
||||
## Frequently Asked Questions:
|
||||
|
||||
**Q: What are the eligibility requirements?**
|
||||
|
||||
A: You must meet all of the following:
|
||||
|
||||
- Pre-seed, Seed or Series A startups (under five years old)
|
||||
- New user of Qdrant Cloud
|
||||
- Has not previously participated in the Qdrant for Startups program
|
||||
- Offer is not valid to existing Qdrant customers
|
||||
- Must be building an AI-driven product or services (agencies or devshops are not eligible)
|
||||
- A live, functional website is required
|
||||
- Billing must be done directly with Qdrant (not through a marketplace)
|
||||
|
||||
**Q: How can I apply to the Qdrant Startup Program?**
|
||||
|
||||
A: Apply through our online form by providing details about your startup and its plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities.
|
||||
|
||||
**Q: What criteria are used to select startups for the program?**
|
||||
|
||||
A: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected.
|
||||
|
||||
**Q: How long is the discount valid, and are there any conditions?**
|
||||
|
||||
A: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount.
|
||||
|
||||
**Q: How can I maximize the co-marketing opportunities offered by the program?**
|
||||
|
||||
A: Engage actively with our marketing team for features on social media, possible appearances in Discord talks or webinars, and case studies to maximize your startup's visibility and showcase your innovative use of Qdrant.
|
||||
|
||||
**Q: Can existing Qdrant customers apply for the Startup Program?**
|
||||
|
||||
A: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered by the program.
|
||||
|
||||
**Q: Can I reapply if my application is initially rejected?**
|
||||
|
||||
A: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply.
|
||||
|
||||
**Q: Who can I contact for more information about the program?**
|
||||
|
||||
A: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com.
|
||||
@@ -130,7 +130,7 @@ I'm really excited to show the power of the Qdrant as vector database. Especiall
|
||||
|
||||
We are happy to welcome this group of people who are deeply committed to advancing vector search technology. We look forward to supporting their vision, and helping them make a bigger impact on the community.
|
||||
|
||||
You can find and chat with them at our [Discord Community](discord.gg/qdrant).
|
||||
You can find and chat with them at our [Discord Community](https://discord.gg/qdrant/).
|
||||
|
||||
### Why become a Qdrant Star?
|
||||
|
||||
|
||||
+2
-2
@@ -35,7 +35,7 @@ Guillaume Marquis, a dedicated Engineer and AI enthusiast, serves as the Chief T
|
||||
|
||||
Who knew that document retrieval could be creative? Guillaume and VirtualBrain help draft sales proposals using past reports. It's fascinating how tech aids deep work beyond basic search tasks.
|
||||
|
||||
Tackling document retrieval and AI assistance, Guillaume furthermore unpacks the ins and outs of searching through vast data using a scoring system, the virtue of RAG for deep work, and going through the 'illusion of work', enhancing insights for knowledge workers while confronting the challenges of scalability and user feedback on hallucinations.
|
||||
Tackling document retrieval and AI assistance, Guillaume furthermore unpacks the ins and outs of searching through vast data using a scoring system, the virtue of [RAG](https://qdrant.tech/rag/rag-evaluation-guide/) for deep work, and going through the 'illusion of work', enhancing insights for knowledge workers while confronting the challenges of scalability and user feedback on hallucinations.
|
||||
|
||||
Here are some key insight from this episode you need to look out for:
|
||||
|
||||
@@ -305,7 +305,7 @@ Guillaume Marquis:
|
||||
So you can trade for free.
|
||||
|
||||
Demetrios:
|
||||
Even better. Look at that, Christmas came early. Well, let's go have some fun, play around with it. And I can't promise, but I may give you some feedback, I may give you some evaluation metrics if it's hallucinating.
|
||||
Even better. Look at that, Christmas came early. Well, let's go have some fun, play around with it. And I can't promise, but I may give you some feedback, I may give you some [evaluation](https://qdrant.tech/rag/rag-evaluation-guide/) metrics if it's hallucinating.
|
||||
|
||||
Guillaume Marquis:
|
||||
Or what if I see some thumbs up or thumbs down, I will know that it's you.
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#Delimiter files are used to separate the list of documentation pages into sections.
|
||||
title: "Examples"
|
||||
type: delimiter
|
||||
weight: 23 # Change this weight to change order of sections
|
||||
weight: 24 # Change this weight to change order of sections
|
||||
sitemapExclude: True
|
||||
_build:
|
||||
publishResources: false
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#Delimiter files are used to separate the list of documentation pages into sections.
|
||||
title: "Support"
|
||||
type: delimiter
|
||||
weight: 26 # Change this weight to change order of sections
|
||||
weight: 27 # Change this weight to change order of sections
|
||||
sitemapExclude: True
|
||||
_build:
|
||||
publishResources: false
|
||||
|
||||
@@ -82,11 +82,11 @@ Here is how you can take a snapshot and recover a collection:
|
||||
- For a single node cluster, call the snapshot endpoint on the exposed URL.
|
||||
- For a multi node cluster call a snapshot on each node of the collection.
|
||||
Specifically, prepend `node-{num}-` to your cluster URL.
|
||||
Then call the [snapshot endpoint](../../concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0.
|
||||
Then call the [snapshot endpoint](/documentation/concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0.
|
||||
- In the response, you'll see the name of the snapshot.
|
||||
2. Delete and recreate the collection.
|
||||
3. Recover the snapshot:
|
||||
- Call the [recover endpoint](../../concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host.
|
||||
- Call the [recover endpoint](/documentation/concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file (`file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host.
|
||||
|
||||
## Backup considerations
|
||||
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
---
|
||||
title: Configure Size & Capacity
|
||||
weight: 40
|
||||
aliases:
|
||||
- capacity
|
||||
---
|
||||
|
||||
# Configuring Qdrant Cloud Cluster Capacity and Size
|
||||
|
||||
We have been asked a lot about the optimal cluster configuration to serve a number of vectors.
|
||||
The only right answer is “It depends”.
|
||||
|
||||
It depends on a number of factors and options you can choose for your collections.
|
||||
|
||||
## Basic configuration
|
||||
|
||||
If you need to keep all vectors in memory for maximum performance, there is a very rough formula for estimating the needed memory size looks like this:
|
||||
|
||||
```text
|
||||
memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5
|
||||
```
|
||||
|
||||
Extra 50% is needed for metadata (indexes, point versions, etc.) as well as for temporary segments constructed during the optimization process.
|
||||
|
||||
If you need to have payloads along with the vectors, it is recommended to store it on the disc, and only keep [indexed fields](../../concepts/indexing/#payload-index) in RAM.
|
||||
Read more about the payload storage in the [Storage](../../concepts/storage/#payload-storage) section.
|
||||
|
||||
|
||||
## Storage focused configuration
|
||||
|
||||
If your priority is to serve large amount of vectors with an average search latency, it is recommended to configure [mmap storage](../../concepts/storage/#configuring-memmap-storage).
|
||||
In this case vectors will be stored on the disc in memory-mapped files, and only the most frequently used vectors will be kept in RAM.
|
||||
|
||||
The amount of available RAM will significantly affect the performance of the search.
|
||||
As a rule of thumb, if you keep 2 times less vectors in RAM, the search latency will be 2 times lower.
|
||||
|
||||
The speed of disks is also important. [Let us know](/documentation/support/) if you have special requirements for a high-volume search.
|
||||
|
||||
## Sub-groups oriented configuration
|
||||
|
||||
|
||||
If your use case assumes that the vectors are split into multiple collections or sub-groups based on payload values,
|
||||
it is recommended to configure memory-map storage.
|
||||
For example, if you serve search for multiple users, but each of them has an subset of vectors which they use independently.
|
||||
|
||||
In this scenario only the active subset of vectors will be kept in RAM, which allows
|
||||
the fast search for the most active and recent users.
|
||||
|
||||
In this case you can estimate required memory size as follows:
|
||||
|
||||
```text
|
||||
memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5
|
||||
```
|
||||
|
||||
## Disk space
|
||||
|
||||
Clusters that support vector search require significant disk space. If you're
|
||||
running low on disk space in your cluster, you can use the UI at
|
||||
[cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster.
|
||||
|
||||
<aside role="status">If you use the Qdrant UI to increase the disk space in your cluster, you
|
||||
cannot decrease that allocation later.</aside>
|
||||
|
||||
If you're running low on disk space, consider the following advantages:
|
||||
|
||||
- Larger Datasets: Supports larger datasets. With vector search,
|
||||
larger datasets can improve the relevance and quality of search results.
|
||||
- Improved Indexing: Supports the use of indexing strategies such as
|
||||
HNSW (Hierarchical Navigable Small World).
|
||||
- Caching: Improves speed when you cache frequently accessed data on disk.
|
||||
- Backups and Redundancy: Allows more frequent backups. Perhaps the most important advantage.
|
||||
@@ -27,11 +27,11 @@ Vertical scaling can be an effective way to improve the performance of a cluster
|
||||
|
||||
In such cases, horizontal scaling may be a more effective solution.
|
||||
|
||||
Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](../../guides/distributed_deployment/#sharding) section for details.
|
||||
Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](/documentation/guides/distributed_deployment/#sharding) section for details.
|
||||
|
||||
After that, you can configure, or change the amount of Qdrant database nodes within a cluster during cluster creation, or on the cluster detail page via "Scale" button.
|
||||
|
||||
Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](../../guides/distributed_deployment/#cluster-scaling).
|
||||
Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](/documentation/guides/distributed_deployment/#cluster-scaling).
|
||||
|
||||
Note, that it is currently not possible to horizontally scale down the cluster in the Qdrant Cloud UI. If you require a horizontal scale down, please open a support ticket.
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@ A free tier cluster only includes 1 single node with the following resources:
|
||||
| Disk space | 4 GB |
|
||||
| Nodes | 1 |
|
||||
|
||||
This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity and sizing](/documentation/cloud/capacity-sizing/).
|
||||
This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity Planning](/documentation/guides/capacity-planning/).
|
||||
|
||||
The choice of cloud providers and regions is limited.
|
||||
|
||||
@@ -73,7 +73,7 @@ This page shows you how to use the Qdrant Cloud Console to create a custom Qdran
|
||||
|
||||
1. Choose your data center region or Hybrid Cloud environment.
|
||||
1. Configure RAM for each node.
|
||||
> For more information, see our [**Capacity and Sizing**](/documentation/cloud/capacity-sizing/) guidance.
|
||||
> For more information, see our [Capacity Planning](/documentation/guides/capacity-planning/) guidance.
|
||||
1. Choose the number of vCPUs per node. If you add more
|
||||
RAM, the menu provides different options for vCPUs.
|
||||
1. Select the number of nodes you want the cluster to be deployed on.
|
||||
|
||||
@@ -28,7 +28,7 @@ These settings can be changed at any time by a corresponding request.
|
||||
|
||||
## Setting up multitenancy
|
||||
|
||||
**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](../../tutorials/multiple-partitions/)
|
||||
**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](/documentation/tutorials/multiple-partitions/)
|
||||
|
||||
**When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise.
|
||||
|
||||
@@ -139,12 +139,12 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
|
||||
In addition to the required options, you can also specify custom values for the following collection options:
|
||||
|
||||
* `hnsw_config` - see [indexing](../indexing/#vector-index) for details.
|
||||
* `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](../storage/#versioning)
|
||||
* `optimizers_config` - see [optimizer](../optimizer/) for details.
|
||||
* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](../../guides/distributed_deployment/#sharding) section for details.
|
||||
* `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details.
|
||||
* `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](/documentation/concepts/storage/#versioning)
|
||||
* `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details.
|
||||
* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](/documentation/guides/distributed_deployment/#sharding) section for details.
|
||||
* `on_disk_payload` - defines where to store payload data. If `true` - payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload.
|
||||
* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details.
|
||||
* `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details.
|
||||
|
||||
Default parameters for the optional collection parameters are defined in [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml).
|
||||
|
||||
@@ -155,7 +155,7 @@ See [schema definitions](https://api.qdrant.tech/api-reference/collections/creat
|
||||
Vectors all live in RAM for very quick access. The `on_disk` parameter can be
|
||||
set in the vector configuration. If true, all vectors will live on disk. This
|
||||
will enable the use of
|
||||
[memmaps](../../concepts/storage/#configuring-memmap-storage),
|
||||
[memmaps](/documentation/concepts/storage/#configuring-memmap-storage),
|
||||
which is suitable for ingesting a large amount of data.
|
||||
|
||||
### Create collection from another collection
|
||||
@@ -466,8 +466,8 @@ For rare use cases, it is possible to create a collection without any vector sto
|
||||
*Available as of v1.1.1*
|
||||
|
||||
For each named vector you can optionally specify
|
||||
[`hnsw_config`](../indexing/#vector-index) or
|
||||
[`quantization_config`](../../guides/quantization/#setting-up-quantization-in-qdrant) to
|
||||
[`hnsw_config`](/documentation/concepts/indexing/#vector-index) or
|
||||
[`quantization_config`](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) to
|
||||
deviate from the collection configuration. This can be useful to fine-tune
|
||||
search performance on a vector level.
|
||||
|
||||
@@ -476,7 +476,7 @@ search performance on a vector level.
|
||||
Vectors all live in RAM for very quick access. On a per-vector basis you can set
|
||||
`on_disk` to true to store all vectors on disk at all times. This will enable
|
||||
the use of
|
||||
[memmaps](../../concepts/storage/#configuring-memmap-storage),
|
||||
[memmaps](/documentation/concepts/storage/#configuring-memmap-storage),
|
||||
which is suitable for ingesting a large amount of data.
|
||||
|
||||
|
||||
@@ -752,7 +752,7 @@ Outside of a unique name, there are no required configuration parameters for spa
|
||||
|
||||
The distance function for sparse vectors is always `Dot` and does not need to be specified.
|
||||
|
||||
However, there are optional parameters to tune the underlying [sparse vector index](../indexing/#sparse-vector-index).
|
||||
However, there are optional parameters to tune the underlying [sparse vector index](/documentation/concepts/indexing/#sparse-vector-index).
|
||||
|
||||
### Check collection existence
|
||||
|
||||
@@ -928,9 +928,9 @@ client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{
|
||||
|
||||
The following parameters can be updated:
|
||||
|
||||
* `optimizers_config` - see [optimizer](../optimizer/) for details.
|
||||
* `hnsw_config` - see [indexing](../indexing/#vector-index) for details.
|
||||
* `quantization_config` - see [quantization](../../guides/quantization/#setting-up-quantization-in-qdrant) for details.
|
||||
* `optimizers_config` - see [optimizer](/documentation/concepts/optimizer/) for details.
|
||||
* `hnsw_config` - see [indexing](/documentation/concepts/indexing/#vector-index) for details.
|
||||
* `quantization_config` - see [quantization](/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details.
|
||||
* `vectors` - vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings.
|
||||
* `params` - other collection parameters, including `write_consistency_factor` and `on_disk_payload`.
|
||||
|
||||
@@ -1495,14 +1495,14 @@ round of automatic optimizations has completed.
|
||||
To clarify: these numbers don't represent the exact amount of points or vectors
|
||||
you have inserted, nor does it represent the exact number of distinguishable
|
||||
points or vectors you can query. If you want to know exact counts, refer to the
|
||||
[count API](../points/#counting-points).
|
||||
[count API](/documentation/concepts/points/#counting-points).
|
||||
|
||||
_Note: these numbers may be removed in a future version of Qdrant._
|
||||
|
||||
### Indexing vectors in HNSW
|
||||
|
||||
In some cases, you might be surprised the value of `indexed_vectors_count` is lower than `vectors_count`. This is an intended behaviour and
|
||||
depends on the [optimizer configuration](../optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the
|
||||
depends on the [optimizer configuration](/documentation/concepts/optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the
|
||||
value of `indexing_threshold`(in kB). If your collection is very small or the dimensionality of the vectors is low, there might be no HNSW segment
|
||||
created and `indexed_vectors_count` might be equal to `0`.
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ aliases:
|
||||
|
||||
# Explore the data
|
||||
|
||||
After mastering the concepts in [search](../search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning.
|
||||
After mastering the concepts in [search](/documentation/concepts/search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning.
|
||||
|
||||
## Recommendation API
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ aliases:
|
||||
# Filtering
|
||||
|
||||
With Qdrant, you can set conditions when searching or retrieving points.
|
||||
For example, you can impose conditions on both the [payload](../payload/) and the `id` of the point.
|
||||
For example, you can impose conditions on both the [payload](/documentation/concepts/payload/) and the `id` of the point.
|
||||
|
||||
Setting additional conditions is important when it is impossible to express all the features of the object in the embedding.
|
||||
Examples include a variety of business requirements: stock availability, user location, or desired price range.
|
||||
@@ -838,7 +838,7 @@ qdrant.NewMatchInt("count", 0)
|
||||
|
||||
The simplest kind of condition is one that checks if the stored value equals the given one.
|
||||
If several values are stored, at least one of them should match the condition.
|
||||
You can apply it to [keyword](../payload/#keyword), [integer](../payload/#integer) and [bool](../payload/#bool) payloads.
|
||||
You can apply it to [keyword](/documentation/concepts/payload/#keyword), [integer](/documentation/concepts/payload/#integer) and [bool](/documentation/concepts/payload/#bool) payloads.
|
||||
|
||||
### Match Any
|
||||
|
||||
@@ -847,7 +847,7 @@ You can apply it to [keyword](../payload/#keyword), [integer](../payload/#intege
|
||||
In case you want to check if the stored value is one of multiple values, you can use the Match Any condition.
|
||||
Match Any works as a logical OR for the given values. It can also be described as a `IN` operator.
|
||||
|
||||
You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads.
|
||||
You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads.
|
||||
|
||||
Example:
|
||||
|
||||
@@ -909,7 +909,7 @@ In case you want to check if the stored value is not one of multiple values, you
|
||||
Match Except works as a logical NOR for the given values.
|
||||
It can also be described as a `NOT IN` operator.
|
||||
|
||||
You can apply it to [keyword](../payload/#keyword) and [integer](../payload/#integer) payloads.
|
||||
You can apply it to [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payloads.
|
||||
|
||||
Example:
|
||||
|
||||
@@ -1908,7 +1908,7 @@ A special case of the `match` condition is the `text` match condition.
|
||||
It allows you to search for a specific substring, token or phrase within the text field.
|
||||
|
||||
Exact texts that will match the condition depend on full-text index configuration.
|
||||
Configuration is defined during the index creation and describe at [full-text index](../indexing/#full-text-index).
|
||||
Configuration is defined during the index creation and describe at [full-text index](/documentation/concepts/indexing/#full-text-index).
|
||||
|
||||
If there is no full-text index for the field, the condition will work as exact substring match.
|
||||
|
||||
@@ -2047,11 +2047,11 @@ Comparisons that can be used:
|
||||
- `lt` - less than
|
||||
- `lte` - less than or equal
|
||||
|
||||
Can be applied to [float](../payload/#float) and [integer](../payload/#integer) payloads.
|
||||
Can be applied to [float](/documentation/concepts/payload/#float) and [integer](/documentation/concepts/payload/#integer) payloads.
|
||||
|
||||
### Datetime Range
|
||||
|
||||
The datetime range is a unique range condition, used for [datetime](../payload/#datetime) payloads, which supports RFC 3339 formats.
|
||||
The datetime range is a unique range condition, used for [datetime](/documentation/concepts/payload/#datetime) payloads, which supports RFC 3339 formats.
|
||||
You do not need to convert dates to UNIX timestaps. During comparison, timestamps are parsed and converted to UTC.
|
||||
|
||||
_Available as of v1.8.0_
|
||||
@@ -2364,7 +2364,7 @@ qdrant.NewGeoRadius("location", 52.520711, 13.403683, 1000.0)
|
||||
It matches with `location`s inside a circle with the `center` at the center and a radius of `radius` meters.
|
||||
|
||||
If several values are stored, at least one of them should match the condition.
|
||||
These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo).
|
||||
These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo).
|
||||
|
||||
#### Geo Polygon
|
||||
Geo Polygons search is useful for when you want to find points inside an irregularly shaped area, for example a country boundary or a forest boundary. A polygon always has an exterior ring and may optionally include interior rings. A lake with an island would be an example of an interior ring. If you wanted to find points in the water but not on the island, you would make an interior ring for the island.
|
||||
@@ -2659,7 +2659,7 @@ qdrant.NewGeoPolygon("location",
|
||||
A match is considered any point location inside or on the boundaries of the given polygon's exterior but not inside any interiors.
|
||||
|
||||
If several location values are stored for a point, then any of them matching will include that point as a candidate in the resultset.
|
||||
These conditions can only be applied to payloads that match the [geo-data format](../payload/#geo).
|
||||
These conditions can only be applied to payloads that match the [geo-data format](/documentation/concepts/payload/#geo).
|
||||
|
||||
### Values count
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ hideInSidebar: false # Optional. If true, the page will not be shown in the side
|
||||
|
||||
*Available as of v1.10.0*
|
||||
|
||||
With the introduction of [many named vectors per point](../vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries,
|
||||
With the introduction of [many named vectors per point](/documentation/concepts/vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries,
|
||||
or by performing the search in more than one stage.
|
||||
|
||||
Qdrant has a flexible and universal interface to make this possible, called `Query API` ([API reference](https://api.qdrant.tech/api-reference/search/query-points)).
|
||||
@@ -793,7 +793,7 @@ Other than the introduction of `prefetch`, the `Query API` has been designed to
|
||||
|
||||
### Query by ID
|
||||
|
||||
Whenever you need to use a vector as an input, you can always use a [point ID](../points/#point-ids) instead.
|
||||
Whenever you need to use a vector as an input, you can always use a [point ID](/documentation/concepts/points/#point-ids) instead.
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
@@ -1397,4 +1397,4 @@ client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{
|
||||
})
|
||||
```
|
||||
|
||||
For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](./search/#search-groups) and [lookup](./search/#lookup-in-groups).
|
||||
For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](/documentation/concepts/search/#search-groups) and [lookup](/documentation/concepts/search/#lookup-in-groups).
|
||||
|
||||
@@ -12,14 +12,14 @@ A key feature of Qdrant is the effective combination of vector and traditional i
|
||||
The indexes in the segments exist independently, but the parameters of the indexes themselves are configured for the whole collection.
|
||||
|
||||
Not all segments automatically have indexes.
|
||||
Their necessity is determined by the [optimizer](../optimizer/) settings and depends, as a rule, on the number of stored points.
|
||||
Their necessity is determined by the [optimizer](/documentation/concepts/optimizer/) settings and depends, as a rule, on the number of stored points.
|
||||
|
||||
## Payload Index
|
||||
|
||||
Payload index in Qdrant is similar to the index in conventional document-oriented databases.
|
||||
This index is built for a specific field and type, and is used for quick point requests by the corresponding filtering condition.
|
||||
|
||||
The index is also used to accurately estimate the filter cardinality, which helps the [query planning](../search/#query-planning) choose a search strategy.
|
||||
The index is also used to accurately estimate the filter cardinality, which helps the [query planning](/documentation/concepts/search/#query-planning) choose a search strategy.
|
||||
|
||||
Creating an index requires additional computational resources and memory, so choosing fields to be indexed is essential. Qdrant does not make this choice but grants it to the user.
|
||||
|
||||
@@ -119,19 +119,19 @@ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection
|
||||
})
|
||||
```
|
||||
|
||||
You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](../filtering/#nested-key).
|
||||
You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](/documentation/concepts/filtering/#nested-key).
|
||||
|
||||
Available field types are:
|
||||
|
||||
* `keyword` - for [keyword](../payload/#keyword) payload, affects [Match](../filtering/#match) filtering conditions.
|
||||
* `integer` - for [integer](../payload/#integer) payload, affects [Match](../filtering/#match) and [Range](../filtering/#range) filtering conditions.
|
||||
* `float` - for [float](../payload/#float) payload, affects [Range](../filtering/#range) filtering conditions.
|
||||
* `bool` - for [bool](../payload/#bool) payload, affects [Match](../filtering/#match) filtering conditions (available as of v1.4.0).
|
||||
* `geo` - for [geo](../payload/#geo) payload, affects [Geo Bounding Box](../filtering/#geo-bounding-box) and [Geo Radius](../filtering/#geo-radius) filtering conditions.
|
||||
* `datetime` - for [datetime](../payload/#datetime) payload, affects [Range](../filtering/#range) filtering conditions (available as of v1.8.0).
|
||||
* `text` - a special kind of index, available for [keyword](../payload/#keyword) / string payloads, affects [Full Text search](../filtering/#full-text-match) filtering conditions.
|
||||
* `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](../payload/#uuid).
|
||||
Affects [Match](../filtering/#match) filtering conditions. (available as of v1.11.0)
|
||||
* `keyword` - for [keyword](/documentation/concepts/payload/#keyword) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions.
|
||||
* `integer` - for [integer](/documentation/concepts/payload/#integer) payload, affects [Match](/documentation/concepts/filtering/#match) and [Range](/documentation/concepts/filtering/#range) filtering conditions.
|
||||
* `float` - for [float](/documentation/concepts/payload/#float) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions.
|
||||
* `bool` - for [bool](/documentation/concepts/payload/#bool) payload, affects [Match](/documentation/concepts/filtering/#match) filtering conditions (available as of v1.4.0).
|
||||
* `geo` - for [geo](/documentation/concepts/payload/#geo) payload, affects [Geo Bounding Box](/documentation/concepts/filtering/#geo-bounding-box) and [Geo Radius](/documentation/concepts/filtering/#geo-radius) filtering conditions.
|
||||
* `datetime` - for [datetime](/documentation/concepts/payload/#datetime) payload, affects [Range](/documentation/concepts/filtering/#range) filtering conditions (available as of v1.8.0).
|
||||
* `text` - a special kind of index, available for [keyword](/documentation/concepts/payload/#keyword) / string payloads, affects [Full Text search](/documentation/concepts/filtering/#full-text-match) filtering conditions.
|
||||
* `uuid` - a special type of index, similar to `keyword`, but optimized for [UUID values](/documentation/concepts/payload/#uuid).
|
||||
Affects [Match](/documentation/concepts/filtering/#match) filtering conditions. (available as of v1.11.0)
|
||||
|
||||
Payload index may occupy some additional memory, so it is recommended to only use index for those fields that are used in filtering conditions.
|
||||
If you need to filter by many fields and the memory limits does not allow to index all of them, it is recommended to choose the field that limits the search result the most.
|
||||
@@ -313,7 +313,7 @@ Available tokenizers are:
|
||||
* `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`.
|
||||
* `multilingual` - special type of tokenizer based on [charabia](https://github.com/meilisearch/charabia) package. It allows proper tokenization and lemmatization for multiple languages, including those with non-latin alphabets and non-space delimiters. See [charabia documentation](https://github.com/meilisearch/charabia) for full list of supported languages supported normalization options. In the default build configuration, qdrant does not include support for all languages, due to the increasing size of the resulting binary. Chinese, Japanese and Korean languages are not enabled by default, but can be enabled by building qdrant from source with `--features multiling-chinese,multiling-japanese,multiling-korean` flags.
|
||||
|
||||
See [Full Text match](../filtering/#full-text-match) for examples of querying with full-text index.
|
||||
See [Full Text match](/documentation/concepts/filtering/#full-text-match) for examples of querying with full-text index.
|
||||
|
||||
### Parameterized index
|
||||
|
||||
@@ -645,7 +645,7 @@ The list will be extended in future versions.
|
||||
|
||||
Many vector search use-cases require multitenancy. In a multi-tenant scenario the collection is expected to contain multiple subsets of data, where each subset belongs to a different tenant.
|
||||
|
||||
Qdrant supports efficient multi-tenant search by enabling [special configuration](../guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant.
|
||||
Qdrant supports efficient multi-tenant search by enabling [special configuration](/documentation/guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant.
|
||||
|
||||
<aside role="note">
|
||||
In Qdrant, tenants are not necessarily non-overlapping. It is possible to have subsets of data that belong to multiple tenants.
|
||||
@@ -960,7 +960,7 @@ storage:
|
||||
|
||||
```
|
||||
|
||||
And so in the process of creating a [collection](../collections/). The `ef` parameter is configured during [the search](../search/) and by default is equal to `ef_construct`.
|
||||
And so in the process of creating a [collection](/documentation/concepts/collections/). The `ef` parameter is configured during [the search](/documentation/concepts/search/) and by default is equal to `ef_construct`.
|
||||
|
||||
HNSW is chosen for several reasons.
|
||||
First, HNSW is well-compatible with the modification that allows Qdrant to use filters during a search.
|
||||
@@ -969,7 +969,7 @@ Second, it is one of the most accurate and fastest algorithms, according to [pub
|
||||
*Available as of v1.1.1*
|
||||
|
||||
The HNSW parameters can also be configured on a collection and named vector
|
||||
level by setting [`hnsw_config`](../indexing/#vector-index) to fine-tune search
|
||||
level by setting [`hnsw_config`](/documentation/concepts/indexing/#vector-index) to fine-tune search
|
||||
performance.
|
||||
|
||||
## Sparse Vector Index
|
||||
|
||||
@@ -9,7 +9,7 @@ aliases:
|
||||
|
||||
It is much more efficient to apply changes in batches than perform each change individually, as many other databases do. Qdrant here is no exception. Since Qdrant operates with data structures that are not always easy to change, it is sometimes necessary to rebuild those structures completely.
|
||||
|
||||
Storage optimization in Qdrant occurs at the segment level (see [storage](../storage/)).
|
||||
Storage optimization in Qdrant occurs at the segment level (see [storage](/documentation/concepts/storage/)).
|
||||
In this case, the segment to be optimized remains readable for the time of the rebuild.
|
||||
|
||||

|
||||
@@ -91,6 +91,6 @@ storage:
|
||||
indexing_threshold_kb: 20000
|
||||
```
|
||||
|
||||
In addition to the configuration file, you can also set optimizer parameters separately for each [collection](../collections/).
|
||||
In addition to the configuration file, you can also set optimizer parameters separately for each [collection](/documentation/concepts/collections/).
|
||||
|
||||
Dynamic parameter updates may be useful, for example, for more efficient initial loading of points. You can disable indexing during the upload process with these settings and enable it immediately after it is finished. As a result, you will not waste extra computation resources on rebuilding the index.
|
||||
@@ -46,11 +46,11 @@ This feature is implemented as additional filters during the search and will ena
|
||||
|
||||
During the filtering, Qdrant will check the conditions over those values that match the type of the filtering condition. If the stored value type does not fit the filtering condition - it will be considered not satisfied.
|
||||
|
||||
For example, you will get an empty output if you apply the [range condition](../filtering/#range) on the string data.
|
||||
For example, you will get an empty output if you apply the [range condition](/documentation/concepts/filtering/#range) on the string data.
|
||||
|
||||
However, arrays (multiple values of the same type) are treated a little bit different. When we apply a filter to an array, it will succeed if at least one of the values inside the array meets the condition.
|
||||
|
||||
The filtering process is discussed in detail in the section [Filtering](../filtering/).
|
||||
The filtering process is discussed in detail in the section [Filtering](/documentation/concepts/filtering/).
|
||||
|
||||
Let's look at the data types that Qdrant supports for searching:
|
||||
|
||||
@@ -1165,7 +1165,7 @@ client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{
|
||||
|
||||
To search more efficiently with filters, Qdrant allows you to create indexes for payload fields by specifying the name and type of field it is intended to be.
|
||||
|
||||
The indexed fields also affect the vector index. See [Indexing](../indexing/) for details.
|
||||
The indexed fields also affect the vector index. See [Indexing](/documentation/concepts/indexing/) for details.
|
||||
|
||||
In practice, we recommend creating an index on those fields that could potentially constrain the results the most.
|
||||
For example, using an index for the object ID will be much more efficient, being unique for each record, than an index by its color, which has only a few possible values.
|
||||
|
||||
@@ -8,7 +8,7 @@ aliases:
|
||||
# Points
|
||||
|
||||
The points are the central entity that Qdrant operates with.
|
||||
A point is a record consisting of a [vector](../vectors/) and an optional [payload](../payload/).
|
||||
A point is a record consisting of a [vector](/documentation/concepts/vectors/) and an optional [payload](/documentation/concepts/payload/).
|
||||
|
||||
It looks like this:
|
||||
|
||||
@@ -21,8 +21,8 @@ It looks like this:
|
||||
}
|
||||
```
|
||||
|
||||
You can search among the points grouped in one [collection](../collections/) based on vector similarity.
|
||||
This procedure is described in more detail in the [search](../search/) and [filtering](../filtering/) sections.
|
||||
You can search among the points grouped in one [collection](/documentation/concepts/collections/) based on vector similarity.
|
||||
This procedure is described in more detail in the [search](/documentation/concepts/search/) and [filtering](/documentation/concepts/filtering/) sections.
|
||||
|
||||
This section explains how to create and manage vectors.
|
||||
|
||||
@@ -343,7 +343,7 @@ Here is a list of supported vector types:
|
||||
It is possible to attach more than one type of vector to a single point.
|
||||
In Qdrant we call it Named Vectors.
|
||||
|
||||
Read more about vector types, how they are stored and optimized in the [vectors](../vectors/) section.
|
||||
Read more about vector types, how they are stored and optimized in the [vectors](/documentation/concepts/vectors/) section.
|
||||
|
||||
|
||||
## Upload points
|
||||
@@ -1424,7 +1424,7 @@ To delete entire points, see [deleting points](#delete-points).
|
||||
|
||||
### Update payload
|
||||
|
||||
Learn how to modify the payload of a point in the [Payload](../payload/#update-payload) section.
|
||||
Learn how to modify the payload of a point in the [Payload](/documentation/concepts/payload/#update-payload) section.
|
||||
|
||||
## Delete points
|
||||
|
||||
@@ -1716,7 +1716,7 @@ Python client:
|
||||
|
||||
Sometimes it might be necessary to get all stored points without knowing ids, or iterate over points that correspond to a filter.
|
||||
|
||||
REST API ([Schema](https://api.qdrant.tech/master/api-reference/search/scroll-points)):
|
||||
REST API ([Schema](https://api.qdrant.tech/master/api-reference/points/scroll-points)):
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/scroll
|
||||
|
||||
@@ -26,13 +26,13 @@ Depending on the `query` parameter, Qdrant might prefer different strategies for
|
||||
| --- | --- |
|
||||
| Nearest Neighbors Search | Vector Similarity Search, also known as k-NN |
|
||||
| Search By Id | Search by an already stored vector - skip embedding model inference |
|
||||
| [Recommendations](../explore/#recommendation-api) | Provide positive and negative examples |
|
||||
| [Discovery Search](../explore/#discovery-api) | Guide the search using context as a one-shot training set |
|
||||
| [Scroll](../points/#scroll-points) | Get all points with optional filtering |
|
||||
| [Grouping](../search/#grouping-api) | Group results by a certain field |
|
||||
| [Order By](../hybrid-queries/#re-ranking-with-stored-values) | Order points by payload key |
|
||||
| [Hybrid Search](../hybrid-queries/#hybrid-search) | Combine multiple queries to get better results |
|
||||
| [Multi-Stage Search](../hybrid-queries/#multi-stage-queries) | Optimize performance for large embeddings |
|
||||
| [Recommendations](/documentation/concepts/explore/#recommendation-api) | Provide positive and negative examples |
|
||||
| [Discovery Search](/documentation/concepts/explore/#discovery-api) | Guide the search using context as a one-shot training set |
|
||||
| [Scroll](/documentation/concepts/points/#scroll-points) | Get all points with optional filtering |
|
||||
| [Grouping](/documentation/concepts/search/#grouping-api) | Group results by a certain field |
|
||||
| [Order By](/documentation/concepts/hybrid-queries/#re-ranking-with-stored-values) | Order points by payload key |
|
||||
| [Hybrid Search](/documentation/concepts/hybrid-queries/#hybrid-search) | Combine multiple queries to get better results |
|
||||
| [Multi-Stage Search](/documentation/concepts/hybrid-queries/#multi-stage-queries) | Optimize performance for large embeddings |
|
||||
| [Random Sampling](#random-sampling) | Get random points from the collection |
|
||||
|
||||
**Nearest Neighbors Search**
|
||||
@@ -406,7 +406,7 @@ Currently, it could be:
|
||||
* `indexed_only` - With this option you can disable the search in those segments where vector index is not built yet. This may be useful if you want to minimize the impact to the search performance whilst the collection is also being updated. Using this option may lead to a partial result if the collection is not fully indexed yet, consider using it only if eventual consistency is acceptable for your use case.
|
||||
|
||||
Since the `filter` parameter is specified, the search is performed only among those points that satisfy the filter condition.
|
||||
See details of possible filters and their work in the [filtering](../filtering/) section.
|
||||
See details of possible filters and their work in the [filtering](/documentation/concepts/filtering/) section.
|
||||
|
||||
Example result of this API would be
|
||||
|
||||
@@ -1282,7 +1282,7 @@ The result of this API contains one array per search requests.
|
||||
|
||||
*Available as of v0.8.3*
|
||||
|
||||
Search and [recommendation](../explore/#recommendation-api) APIs allow to skip first results of the search and return only the result starting from some specified offset:
|
||||
Search and [recommendation](/documentation/concepts/explore/#recommendation-api) APIs allow to skip first results of the search and return only the result starting from some specified offset:
|
||||
|
||||
Example:
|
||||
|
||||
@@ -1423,7 +1423,7 @@ Using an `offset` parameter, will require to internally retrieve `offset + limit
|
||||
|
||||
It is possible to group results by a certain field. This is useful when you have multiple points for the same item, and you want to avoid redundancy of the same item in the results.
|
||||
|
||||
For example, if you have a large document split into multiple chunks, and you want to search or [recommend](../explore/#recommendation-api) on a per-document basis, you can group the results by the document ID.
|
||||
For example, if you have a large document split into multiple chunks, and you want to search or [recommend](/documentation/concepts/explore/#recommendation-api) on a per-document basis, you can group the results by the document ID.
|
||||
|
||||
Consider having points with the following payloads:
|
||||
|
||||
@@ -1631,7 +1631,7 @@ If the `group_by` field of a point is an array (e.g. `"document_id": ["a", "b"]`
|
||||
|
||||
**Limitations**:
|
||||
|
||||
* Only [keyword](../payload/#keyword) and [integer](../payload/#integer) payload values are supported for the `group_by` parameter. Payload values with other types will be ignored.
|
||||
* Only [keyword](/documentation/concepts/payload/#keyword) and [integer](/documentation/concepts/payload/#integer) payload values are supported for the `group_by` parameter. Payload values with other types will be ignored.
|
||||
* At the moment, pagination is not enabled when using **groups**, so the `offset` parameter is not allowed.
|
||||
|
||||
### Lookup in groups
|
||||
@@ -1964,10 +1964,10 @@ This process is called query planning.
|
||||
The strategy selection process relies heavily on heuristics and can vary from release to release.
|
||||
However, the general principles are:
|
||||
|
||||
* planning is performed for each segment independently (see [storage](../storage/) for more information about segments)
|
||||
* planning is performed for each segment independently (see [storage](/documentation/concepts/storage/) for more information about segments)
|
||||
* prefer a full scan if the amount of points is below a threshold
|
||||
* estimate the cardinality of a filtered result before selecting a strategy
|
||||
* retrieve points using payload index (see [indexing](../indexing/)) if cardinality is below threshold
|
||||
* retrieve points using payload index (see [indexing](/documentation/concepts/indexing/)) if cardinality is below threshold
|
||||
* use filterable vector index if the cardinality is above a threshold
|
||||
|
||||
You can adjust the threshold using a [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), as well as independently for each collection.
|
||||
|
||||
@@ -612,7 +612,7 @@ also configure to use an [S3 storage](#s3) service for them.
|
||||
By default, snapshots are stored at `./snapshots` or at `/qdrant/snapshots` when
|
||||
using our Docker image.
|
||||
|
||||
The target directory can be controlled through the [configuration](../../guides/configuration/):
|
||||
The target directory can be controlled through the [configuration](/documentation/guides/configuration/):
|
||||
|
||||
```yaml
|
||||
storage:
|
||||
@@ -640,7 +640,7 @@ storage:
|
||||
|
||||
Rather than storing snapshots on the local file system, you may also configure
|
||||
to store snapshots in an S3-compatible storage service. To enable this, you must
|
||||
configure it in the [configuration](../../guides/configuration/) file.
|
||||
configure it in the [configuration](/documentation/guides/configuration/) file.
|
||||
|
||||
For example, to configure for AWS S3:
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ Each segment has its independent vector and payload storage as well as indexes.
|
||||
Data stored in segments usually do not overlap.
|
||||
However, storing the same point in different segments will not cause problems since the search contains a deduplication mechanism.
|
||||
|
||||
The segments consist of vector and payload storages, vector and payload [indexes](../indexing/), and id mapper, which stores the relationship between internal and external ids.
|
||||
The segments consist of vector and payload storages, vector and payload [indexes](/documentation/concepts/indexing/), and id mapper, which stores the relationship between internal and external ids.
|
||||
|
||||
A segment can be `appendable` or `non-appendable` depending on the type of storage and index used.
|
||||
You can freely add, delete and query data in the `appendable` segment.
|
||||
@@ -161,8 +161,8 @@ This is the recommended way, in case your Qdrant instance operates with fast dis
|
||||
|
||||
There are two ways to do this:
|
||||
|
||||
1. You can set the threshold globally in the [configuration file](../../guides/configuration/). The parameter is called `memmap_threshold_kb`.
|
||||
2. You can set the threshold for each collection separately during [creation](../collections/#create-collection) or [update](../collections/#update-collection-parameters).
|
||||
1. You can set the threshold globally in the [configuration file](/documentation/guides/configuration/). The parameter is called `memmap_threshold_kb`.
|
||||
2. You can set the threshold for each collection separately during [creation](/documentation/concepts/collections/#create-collection) or [update](/documentation/concepts/collections/#update-collection-parameters).
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -295,7 +295,7 @@ The rule of thumb to set the memmap threshold parameter is simple:
|
||||
- if you have a high write load and low RAM - set memmap threshold lower than `indexing_threshold` to e.g. 10000. In this case the optimizer will convert the segments to memmap storage first and will only apply indexing after that.
|
||||
|
||||
In addition, you can use memmap storage not only for vectors, but also for HNSW index.
|
||||
To enable this, you need to set the `hnsw_config.on_disk` parameter to `true` during collection [creation](../collections/#create-a-collection) or [updating](../collections/#update-collection-parameters).
|
||||
To enable this, you need to set the `hnsw_config.on_disk` parameter to `true` during collection [creation](/documentation/concepts/collections/#create-a-collection) or [updating](/documentation/concepts/collections/#update-collection-parameters).
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -452,7 +452,7 @@ If you need to query vectors with some payload-based conditions - checking value
|
||||
In this scenario, we recommend creating a payload index for each field used in filtering conditions to avoid disk access.
|
||||
Once you create the field index, Qdrant will preserve all values of the indexed field in RAM regardless of the payload storage type.
|
||||
|
||||
You can specify the desired type of payload storage with [configuration file](../../guides/configuration/) or with collection parameter `on_disk_payload` during [creation](../collections/#create-collection) of the collection.
|
||||
You can specify the desired type of payload storage with [configuration file](/documentation/guides/configuration/) or with collection parameter `on_disk_payload` during [creation](/documentation/concepts/collections/#create-collection) of the collection.
|
||||
|
||||
## Versioning
|
||||
|
||||
|
||||
@@ -1409,7 +1409,7 @@ client.create_collection(
|
||||
),
|
||||
sparse_vectors_config={
|
||||
"text": models.SparseVectorParams(
|
||||
index=models.SparseIndexConfig(datatype=models.Datatype.FLOAT16)
|
||||
index=models.SparseIndexParams(datatype=models.Datatype.FLOAT16)
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
@@ -7,15 +7,14 @@ weight: 18
|
||||
|
||||
| Integration | Description |
|
||||
| ------------------------------- | -------------------------------------------------------------------------------------------------- |
|
||||
| [Airbyte](./airbyte/) | Data integration platform specialising in ELT pipelines. |
|
||||
| [Airflow](./airflow/) | Platform designed for developing, scheduling, and monitoring batch-oriented workflows. |
|
||||
| [Connect](./redpanda/) | Declarative data-agnostic streaming service for efficient, stateless processing. |
|
||||
| [Confluent](./confluent/) | Fully-managed data streaming platform with a cloud-native Apache Kafka engine. |
|
||||
| [DLT](./dlt/) | Python library to simplify data loading processes between several sources and destinations. |
|
||||
| [Fluvio](./fluvio/) | Rust-based platform for high speed, real-time data processing. |
|
||||
| [Fondant](./fondant/) | Framework for developing datasets, sharing reusable operations and data processing trees. |
|
||||
| [MindsDB](./mindsdb/) | Platform to deploy, serve, and fine-tune models with numerous data source integrations. |
|
||||
| [NiFi](./nifi/) | Data ingestion platform to manage data transfer between different sources and destination systems. |
|
||||
| [Spark](./spark/) | A unified analytics engine for large-scale data processing. |
|
||||
| [Unstructured](./unstructured/) | Python library with components for ingesting and pre-processing data from numerous sources. |
|
||||
|
||||
| [Airbyte](/documentation/data-management/airbyte/) | Data integration platform specialising in ELT pipelines. |
|
||||
| [Airflow](/documentation/data-management/airflow/) | Platform designed for developing, scheduling, and monitoring batch-oriented workflows. |
|
||||
| [Connect](/documentation/data-management/redpanda/) | Declarative data-agnostic streaming service for efficient, stateless processing. |
|
||||
| [Confluent](/documentation/data-management/confluent/) | Fully-managed data streaming platform with a cloud-native Apache Kafka engine. |
|
||||
| [DLT](/documentation/data-management/dlt/) | Python library to simplify data loading processes between several sources and destinations. |
|
||||
| [Fluvio](/documentation/data-management/fluvio/) | Rust-based platform for high speed, real-time data processing. |
|
||||
| [Fondant](/documentation/data-management/fondant/) | Framework for developing datasets, sharing reusable operations and data processing trees. |
|
||||
| [MindsDB](/documentation/data-management/mindsdb/) | Platform to deploy, serve, and fine-tune models with numerous data source integrations. |
|
||||
| [NiFi](/documentation/data-management/nifi/) | Data ingestion platform to manage data transfer between different sources and destination systems. |
|
||||
| [Spark](/documentation/data-management/spark/) | A unified analytics engine for large-scale data processing. |
|
||||
| [Unstructured](/documentation/data-management/unstructured/) | Python library with components for ingesting and pre-processing data from numerous sources. |
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: Practice Datasets
|
||||
weight: 28
|
||||
weight: 29
|
||||
---
|
||||
|
||||
# Common Datasets in Snapshot Format
|
||||
|
||||
@@ -17,31 +17,19 @@ Additionally, [any open-source embeddings from HuggingFace](https://huggingface.
|
||||
|
||||
| Embeddings Providers | Description |
|
||||
| ----------------------------- | ----------- |
|
||||
| [Aleph Alpha](./aleph-alpha/) | Multilingual embeddings focused on European languages. |
|
||||
| [Azure](./azure/) | Microsoft's embedding model selection. |
|
||||
| [Bedrock](./bedrock/) | AWS managed service for foundation models and embeddings. |
|
||||
| [Clarifai](./clarifai/) | Embeddings for image and video recognition. |
|
||||
| [Clip](./clip/) | Aligns images and text, created by OpenAI. |
|
||||
| [Cohere](./cohere/) | Language model embeddings for NLP tasks. |
|
||||
| [Databricks](./databricks/) | Scalable embeddings integrated with Apache Spark. |
|
||||
| [Gemini](./gemini/) | Google’s multimodal embeddings for text and vision. |
|
||||
| [GPT4All](./gpt4all/) | Open-source, local embeddings for privacy-focused use. |
|
||||
| [GradientAI](./gradient/) | AI Models for custom enterprise tasks.|
|
||||
| [Instruct](./instruct/) | Embeddings tuned for following instructions. |
|
||||
| [Jina AI](./jina-embeddings/) | Customizable embeddings for neural search. |
|
||||
| [John Snow Labs](./johnsnow/) | Medical and clinical embeddings. |
|
||||
| [Mistral](./mistral/) | Open-source, efficient language model embeddings. |
|
||||
| [MixedBread](./mixedbread/) | Lightweight embeddings for constrained environments. |
|
||||
| [Mixpeek](./mixpeek/) | Managed SDK for video chunking, embedding, and post-processing. |
|
||||
| [Nomic](./nomic/) | Embeddings for data visualization. |
|
||||
| [Nvidia](./nvidia/) | GPU-optimized embeddings from Nvidia. |
|
||||
| [OCI](./oci/) | Oracle Cloud’s AI service with embeddings. |
|
||||
| [Ollama](./ollama/) | Embeddings for conversational AI. |
|
||||
| [OpenAI](./openai/) | Industry-leading embeddings for NLP. |
|
||||
| [OpenCLIP](./openclip/) | OS implementation of CLIP for image and text. |
|
||||
| [Prem AI](./premai/) | Precise language embeddings. |
|
||||
| [Snowflake](./snowflake/) | Scalable embeddings for big data. |
|
||||
| [Together AI](./togetherai/) | Community-driven, open-source embeddings. |
|
||||
| [Upstage](./upstage/) | Embeddings for speech and language tasks. |
|
||||
| [Voyage AI](./voyage/) | Navigation and spatial understanding embeddings. |
|
||||
| [Watsonx](./watsonx/) | IBM's enterprise-grade embeddings. |
|
||||
| [Aleph Alpha](/documentation/embeddings/aleph-alpha/) | Multilingual embeddings focused on European languages. |
|
||||
| [Bedrock](/documentation/embeddings/bedrock/) | AWS managed service for foundation models and embeddings. |
|
||||
| [Cohere](/documentation/embeddings/cohere/) | Language model embeddings for NLP tasks. |
|
||||
| [Gemini](/documentation/embeddings/gemini/) | Google’s multimodal embeddings for text and vision.
|
||||
| [Jina AI](/documentation/embeddings/jina-embeddings/) | Customizable embeddings for neural search. |
|
||||
| [Mistral](/documentation/embeddings/mistral/) | Open-source, efficient language model embeddings. |
|
||||
| [MixedBread](/documentation/embeddings/mixedbread/) | Lightweight embeddings for constrained environments. |
|
||||
| [Mixpeek](/documentation/embeddings/mixpeek/) | Managed SDK for video chunking, embedding, and post-processing. |
|
||||
| [Nomic](/documentation/embeddings/nomic/) | Embeddings for data visualization. |
|
||||
| [Nvidia](/documentation/embeddings/nvidia/) | GPU-optimized embeddings from Nvidia. |
|
||||
| [Ollama](/documentation/embeddings/ollama/) | Embeddings for conversational AI. |
|
||||
| [OpenAI](/documentation/embeddings/openai/) | Industry-leading embeddings for NLP. |
|
||||
| [Prem AI](/documentation/embeddings/premai/) | Precise language embeddings. |
|
||||
| [Snowflake](/documentation/embeddings/snowflake/) | Scalable embeddings for big data. |
|
||||
| [Upstage](/documentation/embeddings/upstage/) | Embeddings for speech and language tasks. |
|
||||
| [Voyage AI](/documentation/embeddings/voyage/) | Navigation and spatial understanding embeddings. |
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
---
|
||||
title: Azure OpenAI
|
||||
weight: 950
|
||||
---
|
||||
|
||||
# Using Azure OpenAI with Qdrant
|
||||
|
||||
Azure OpenAI is Microsoft's platform for AI embeddings, focusing on powerful text and data analytics. These embeddings are suitable for high-precision vector searches in Qdrant.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required packages using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install openai azure-identity python-dotenv qdrant-client
|
||||
```
|
||||
|
||||
## Code Example
|
||||
|
||||
```python
|
||||
import os
|
||||
import openai
|
||||
import dotenv
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from azure.identity import DefaultAzureCredential, get_bearer_token_provider
|
||||
|
||||
dotenv.load_dotenv()
|
||||
|
||||
# Set to True if using Azure Active Directory for authentication
|
||||
use_azure_active_directory = False
|
||||
|
||||
# Qdrant client setup
|
||||
qdrant_client = qdrant_client.QdrantClient(url="http://localhost:6333")
|
||||
|
||||
# Azure OpenAI Authentication
|
||||
if not use_azure_active_directory:
|
||||
endpoint = os.environ["AZURE_OPENAI_ENDPOINT"]
|
||||
api_key = os.environ["AZURE_OPENAI_API_KEY"]
|
||||
|
||||
client = openai.AzureOpenAI(
|
||||
azure_endpoint=endpoint,
|
||||
api_key=api_key,
|
||||
api_version="2023-09-01-preview"
|
||||
)
|
||||
else:
|
||||
endpoint = os.environ["AZURE_OPENAI_ENDPOINT"]
|
||||
client = openai.AzureOpenAI(
|
||||
azure_endpoint=endpoint,
|
||||
azure_ad_token_provider=get_bearer_token_provider(DefaultAzureCredential(), "https://cognitiveservices.azure.com/.default"),
|
||||
api_version="2023-09-01-preview"
|
||||
)
|
||||
|
||||
# Deployment name of the model in Azure OpenAI Studio
|
||||
deployment = "your-deployment-name" # Replace with your deployment name
|
||||
|
||||
# Generate embeddings using the Azure OpenAI client
|
||||
text_input = "The food was delicious and the waiter..."
|
||||
embeddings_response = client.embeddings.create(
|
||||
model=deployment,
|
||||
input=text_input
|
||||
)
|
||||
|
||||
# Extract the embedding vector from the response
|
||||
embedding_vector = embeddings_response.data[0].embedding
|
||||
|
||||
# Insert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="MyCollection",
|
||||
points=Batch(
|
||||
ids=[1], # This ID can be dynamically assigned or managed
|
||||
vectors=[embedding_vector],
|
||||
)
|
||||
)
|
||||
|
||||
print("Embedding successfully upserted into Qdrant.")
|
||||
```
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
title: Clarifai
|
||||
weight: 1200
|
||||
---
|
||||
|
||||
# Using Clarifai Embeddings with Qdrant
|
||||
|
||||
Clarifai is a leading provider of visual embeddings, which are particularly strong in image and video analysis. Clarifai offers an API that allows you to create embeddings for various media types, which can be integrated into Qdrant for efficient vector search and retrieval.
|
||||
|
||||
You can install the Clarifai Python client with pip:
|
||||
|
||||
```bash
|
||||
pip install clarifai-client
|
||||
```
|
||||
|
||||
## Integration Example
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from clarifai.rest import ClarifaiApp
|
||||
|
||||
# Initialize Clarifai client
|
||||
clarifai_app = ClarifaiApp(api_key="<< your_api_key >>")
|
||||
|
||||
# Choose the model for embeddings
|
||||
model = clarifai_app.public_models.general_embedding_model
|
||||
|
||||
# Upload and get embeddings for an image
|
||||
image_path = "./path/to/the/image.jpg"
|
||||
response = model.predict_by_filename(image_path)
|
||||
|
||||
# Extract the embedding from the response
|
||||
embedding = response['outputs'][0]['data']['embeddings'][0]['vector']
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient()
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="MyCollection",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embedding],
|
||||
)
|
||||
)
|
||||
```
|
||||
@@ -1,50 +0,0 @@
|
||||
---
|
||||
title: Clip
|
||||
weight: 1300
|
||||
---
|
||||
|
||||
# Using Clip with Qdrant
|
||||
|
||||
CLIP (Contrastive Language-Image Pre-Training) provides advanced AI capabilities including natural language processing and computer vision. CLIP is a neural network trained on a variety of (image, text) pairs. It can be instructed in natural language to predict the most relevant text snippet, given an image, without directly optimizing for the task, similarly to the zero-shot capabilities of GPT-2 and 3.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install clip-client
|
||||
```
|
||||
## Integration Example
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from transformers import CLIPProcessor, CLIPModel
|
||||
from PIL import Image
|
||||
|
||||
# Load the CLIP model and processor
|
||||
model = CLIPModel.from_pretrained("openai/clip-vit-base-patch32")
|
||||
processor = CLIPProcessor.from_pretrained("openai/clip-vit-base-patch32")
|
||||
|
||||
# Load and process the image
|
||||
image = Image.open("path/to/image.jpg")
|
||||
inputs = processor(images=image, return_tensors="pt")
|
||||
|
||||
# Generate embeddings
|
||||
with torch.no_grad():
|
||||
embeddings = model.get_image_features(**inputs).numpy().tolist()
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="ImageEmbeddings",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=embeddings,
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
@@ -1,37 +0,0 @@
|
||||
---
|
||||
title: Databricks Embeddings
|
||||
weight: 1500
|
||||
---
|
||||
|
||||
# Using Databricks Embeddings with Qdrant
|
||||
|
||||
Databricks offers an advanced platform for generating embeddings, especially within large-scale data environments. You can use the following Python code to integrate Databricks-generated embeddings with Qdrant.
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from databricks import sql
|
||||
|
||||
# Connect to Databricks SQL endpoint
|
||||
connection = sql.connect(server_hostname='your_hostname',
|
||||
http_path='your_http_path',
|
||||
access_token='your_access_token')
|
||||
|
||||
# Execute a query to get embeddings
|
||||
query = "SELECT embedding FROM your_table WHERE id = 1"
|
||||
cursor = connection.cursor()
|
||||
cursor.execute(query)
|
||||
embedding = cursor.fetchone()[0]
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="DatabricksEmbeddings",
|
||||
points=Batch(
|
||||
ids=[1], # Unique ID for the data point
|
||||
vectors=[embedding], # Embedding fetched from Databricks
|
||||
)
|
||||
)
|
||||
```
|
||||
@@ -1,47 +0,0 @@
|
||||
---
|
||||
title: GPT4All
|
||||
weight: 1700
|
||||
---
|
||||
|
||||
# Using GPT4All with Qdrant
|
||||
|
||||
GPT4All offers a range of large language models that can be fine-tuned for various applications. GPT4All runs large language models (LLMs) privately on everyday desktops & laptops.
|
||||
|
||||
No API calls or GPUs required - you can just download the application and get started. Use GPT4All in Python to program with LLMs implemented with the llama.cpp backend and Nomic's C backend.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install gpt4all
|
||||
```
|
||||
|
||||
Here is how you might connect to GPT4ALL using Qdrant:
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from gpt4all import GPT4All
|
||||
|
||||
# Initialize GPT4All model
|
||||
model = GPT4All("gpt4all-lora-quantized")
|
||||
|
||||
# Generate embeddings for a text
|
||||
text = "GPT4All enables open-source AI applications."
|
||||
embeddings = model.embed(text)
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="OpenSourceAI",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
---
|
||||
title: GradientAI
|
||||
weight: 1750
|
||||
---
|
||||
|
||||
# Using GradientAI with Qdrant
|
||||
|
||||
GradientAI provides state-of-the-art models for generating embeddings, which are highly effective for vector search tasks in Qdrant.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required packages using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install gradientai python-dotenv qdrant-client
|
||||
```
|
||||
|
||||
## Code Example
|
||||
|
||||
```python
|
||||
from dotenv import load_dotenv
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from gradientai import Gradient
|
||||
|
||||
load_dotenv()
|
||||
|
||||
def main() -> None:
|
||||
# Initialize GradientAI client
|
||||
gradient = Gradient()
|
||||
|
||||
# Retrieve the embeddings model
|
||||
embeddings_model = gradient.get_embeddings_model(slug="bge-large")
|
||||
|
||||
# Generate embeddings for your data
|
||||
generate_embeddings_response = embeddings_model.generate_embeddings(
|
||||
inputs=[
|
||||
"Multimodal brain MRI is the preferred method to evaluate for acute ischemic infarct and ideally should be obtained within 24 hours of symptom onset, and in most centers will follow a NCCT",
|
||||
"CTA has a higher sensitivity and positive predictive value than magnetic resonance angiography (MRA) for detection of intracranial stenosis and occlusion and is recommended over time-of-flight (without contrast) MRA",
|
||||
"Echocardiographic strain imaging has the advantage of detecting early cardiac involvement, even before thickened walls or symptoms are apparent",
|
||||
],
|
||||
)
|
||||
|
||||
# Initialize Qdrant client
|
||||
client = qdrant_client.QdrantClient(url="http://localhost:6333")
|
||||
|
||||
# Upsert the embeddings into Qdrant
|
||||
for i, embedding in enumerate(generate_embeddings_response.embeddings):
|
||||
client.upsert(
|
||||
collection_name="MedicalRecords",
|
||||
points=Batch(
|
||||
ids=[i + 1], # Unique ID for each embedding
|
||||
vectors=[embedding.embedding],
|
||||
)
|
||||
)
|
||||
|
||||
print("Embeddings successfully upserted into Qdrant.")
|
||||
gradient.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
```
|
||||
@@ -1,42 +0,0 @@
|
||||
---
|
||||
title: Instruct
|
||||
weight: 1800
|
||||
---
|
||||
|
||||
# Using Instruct with Qdrant
|
||||
|
||||
Instruct is a specialized provider offering detailed embeddings for instructional content, which can be effectively used with Qdrant. With Instruct every text input is embedded together with instructions explaining the use case (e.g., task and domain descriptions). Unlike encoders from prior work that are more specialized, INSTRUCTOR is a single embedder that can generate text embeddings tailored to different downstream tasks and domains, without any further training.
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install instruct
|
||||
```
|
||||
|
||||
Below is an example of how to obtain embeddings using Instruct's API and store them in a Qdrant collection:
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from instruct import Instruct
|
||||
|
||||
# Initialize Instruct model
|
||||
model = Instruct("instruct-base")
|
||||
|
||||
# Generate embeddings for instructional content
|
||||
text = "Instruct provides detailed embeddings for learning content."
|
||||
embeddings = model.embed(text)
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="LearningContent",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
@@ -1,49 +0,0 @@
|
||||
---
|
||||
title: John Snow Labs
|
||||
weight: 2000
|
||||
---
|
||||
|
||||
# Using John Snow Labs with Qdrant
|
||||
|
||||
John Snow Labs offers a variety of models, particularly in the healthcare domain. They have pre-trained models that can generate embeddings for medical text data.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install johnsnowlabs
|
||||
```
|
||||
|
||||
|
||||
Here is an example of how you might obtain embeddings using John Snow Labs's API and store them in a Qdrant collection:
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from johnsnowlabs import nlp
|
||||
|
||||
# Load the pre-trained model, for example, a named entity recognition (NER) model
|
||||
model = nlp.load_model("ner_jsl")
|
||||
|
||||
# Sample text to generate embeddings
|
||||
text = "John Snow Labs provides state-of-the-art healthcare NLP solutions."
|
||||
|
||||
# Generate embeddings for the text
|
||||
document = nlp.DocumentAssembler().setInput(text)
|
||||
embeddings = model.transform(document).collectEmbeddings()
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embeddings into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="HealthcareNLP",
|
||||
points=Batch(
|
||||
ids=[1], # This would be your unique ID for the data point
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
---
|
||||
title: OCI (Oracle Cloud Infrastructure)
|
||||
weight: 2500
|
||||
---
|
||||
|
||||
# Using OCI (Oracle Cloud Infrastructure) with Qdrant
|
||||
|
||||
OCI provides robust cloud-based embeddings for various media types. The Generative AI Embedding Models convert textual input - ranging from phrases and sentences to entire paragraphs - into a structured format known as embeddings. Each piece of text input is transformed into a numerical array consisting of 1024 distinct numbers.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install oci
|
||||
```
|
||||
|
||||
## Code Example
|
||||
|
||||
Below is an example of how to obtain embeddings using OCI (Oracle Cloud Infrastructure)'s API and store them in a Qdrant collection:
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
import oci
|
||||
|
||||
# Initialize OCI client
|
||||
config = oci.config.from_file()
|
||||
ai_client = oci.ai_language.AIServiceLanguageClient(config)
|
||||
|
||||
# Generate embeddings using OCI's AI service
|
||||
text = "OCI provides cloud-based AI services."
|
||||
response = ai_client.batch_detect_language_entities(text)
|
||||
embeddings = response.data[0].entities[0].embedding
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="CloudAI",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
@@ -1,41 +0,0 @@
|
||||
---
|
||||
title: OpenCLIP
|
||||
weight: 2750
|
||||
---
|
||||
|
||||
# Using OpenCLIP with Qdrant
|
||||
|
||||
OpenCLIP is an open-source implementation of the CLIP model, allowing for open source generation of multimodal embeddings that link text and images.
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
import open_clip
|
||||
|
||||
# Load the OpenCLIP model and tokenizer
|
||||
model, preprocess = open_clip.create_model_and_transforms('ViT-B-32', pretrained='openai')
|
||||
tokenizer = open_clip.get_tokenizer('ViT-B-32')
|
||||
|
||||
# Generate embeddings for a text
|
||||
text = "A photo of a cat"
|
||||
text_inputs = tokenizer([text])
|
||||
|
||||
with torch.no_grad():
|
||||
text_features = model.encode_text(text_inputs)
|
||||
|
||||
# Convert tensor to a list
|
||||
embeddings = text_features[0].cpu().numpy().tolist()
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="OpenCLIPEmbeddings",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
```
|
||||
|
||||
@@ -1,43 +0,0 @@
|
||||
---
|
||||
title: Together AI
|
||||
weight: 3000
|
||||
---
|
||||
|
||||
# Using Together AI with Qdrant
|
||||
|
||||
Together AI focuses on collaborative AI embeddings that enhance multi-user search scenarios when integrated with Qdrant.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install togetherai
|
||||
```
|
||||
## Integration Example
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from togetherai import TogetherAI
|
||||
|
||||
# Initialize Together AI model
|
||||
model = TogetherAI("togetherai-collab")
|
||||
|
||||
# Generate embeddings for collaborative content
|
||||
text = "Together AI enhances collaborative content search."
|
||||
embeddings = model.embed(text)
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="CollaborativeContent",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
@@ -1,50 +0,0 @@
|
||||
|
||||
---
|
||||
title: Watsonx
|
||||
weight: 3000
|
||||
aliases:
|
||||
- /documentation/examples/watsonx-search/
|
||||
- /documentation/tutorials/watsonx-search/
|
||||
- /documentation/integrations/watsonx/
|
||||
---
|
||||
|
||||
# Using Watsonx with Qdrant
|
||||
|
||||
Watsonx is IBM's platform for AI embeddings, focusing on enterprise-level text and data analytics. These embeddings are suitable for high-precision vector searches in Qdrant.
|
||||
|
||||
## Installation
|
||||
|
||||
You can install the required package using the following pip command:
|
||||
|
||||
```bash
|
||||
pip install watsonx
|
||||
```
|
||||
|
||||
## Code Example
|
||||
|
||||
|
||||
```python
|
||||
import qdrant_client
|
||||
from qdrant_client.models import Batch
|
||||
from watsonx import Watsonx
|
||||
|
||||
# Initialize Watsonx AI model
|
||||
model = Watsonx("watsonx-model")
|
||||
|
||||
# Generate embeddings for enterprise data
|
||||
text = "Watsonx provides enterprise-level NLP solutions."
|
||||
embeddings = model.embed(text)
|
||||
|
||||
# Initialize Qdrant client
|
||||
qdrant_client = qdrant_client.QdrantClient(host="localhost", port=6333)
|
||||
|
||||
# Upsert the embedding into Qdrant
|
||||
qdrant_client.upsert(
|
||||
collection_name="EnterpriseData",
|
||||
points=Batch(
|
||||
ids=[1],
|
||||
vectors=[embeddings],
|
||||
)
|
||||
)
|
||||
|
||||
```
|
||||
@@ -1,21 +1,21 @@
|
||||
---
|
||||
title: Build Prototypes
|
||||
weight: 25
|
||||
weight: 26
|
||||
---
|
||||
# Examples
|
||||
|
||||
| End-to-End Code Samples | Description | Stack |
|
||||
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
|
||||
| [Multitenancy with LlamaIndex](../examples/llama-index-multitenancy/) | Handle data coming from multiple users in LlamaIndex. | Qdrant, Python, LlamaIndex |
|
||||
| [Implement custom connector for Cohere RAG](../examples/cohere-rag-connector/) | Bring data stored in Qdrant to Cohere RAG | Qdrant, Cohere, FastAPI |
|
||||
| [Chatbot for Interactive Learning](../examples/rag-chatbot-red-hat-openshift-haystack/) | Build a Private RAG Chatbot for Interactive Learning | Qdrant, Haystack, OpenShift |
|
||||
| [Information Extraction Engine](../examples/rag-chatbot-vultr-dspy-ollama/) | Build a Private RAG Information Extraction Engine | Qdrant, Vultr, DSPy, Ollama |
|
||||
| [System for Employee Onboarding](../examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/) | Build a RAG System for Employee Onboarding | Qdrant, Cohere, LangChain |
|
||||
| [System for Contract Management](../examples/rag-contract-management-stackit-aleph-alpha/) | Build a Region-Specific RAG System for Contract Management | Qdrant, Aleph Alpha, STACKIT |
|
||||
| [Question-Answering System for Customer Support](../examples/rag-customer-support-cohere-airbyte-aws/) | Build a RAG System for AI Customer Support | Qdrant, Cohere, Airbyte, AWS |
|
||||
| [Hybrid Search on PDF Documents](../examples/hybrid-search-llamaindex-jinaai/) | Develop a Hybrid Search System for Product PDF Manuals | Qdrant, LlamaIndex, Jina AI
|
||||
| [Blog-Reading RAG Chatbot](../examples/rag-chatbot-scaleway) | Develop a RAG-based Chatbot on Scaleway and with LangChain | Qdrant, LangChain, GPT-4o
|
||||
| [Movie Recommendation System](../examples/recommendation-system-ovhcloud/) | Build a Movie Recommendation System with LlamaIndex and With JinaAI | Qdrant |
|
||||
| [Multitenancy with LlamaIndex](/documentation/examples/llama-index-multitenancy/) | Handle data coming from multiple users in LlamaIndex. | Qdrant, Python, LlamaIndex |
|
||||
| [Implement custom connector for Cohere RAG](/documentation/examples/cohere-rag-connector/) | Bring data stored in Qdrant to Cohere RAG | Qdrant, Cohere, FastAPI |
|
||||
| [Chatbot for Interactive Learning](/documentation/examples/rag-chatbot-red-hat-openshift-haystack/) | Build a Private RAG Chatbot for Interactive Learning | Qdrant, Haystack, OpenShift |
|
||||
| [Information Extraction Engine](/documentation/examples/rag-chatbot-vultr-dspy-ollama/) | Build a Private RAG Information Extraction Engine | Qdrant, Vultr, DSPy, Ollama |
|
||||
| [System for Employee Onboarding](/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/) | Build a RAG System for Employee Onboarding | Qdrant, Cohere, LangChain |
|
||||
| [System for Contract Management](/documentation/examples/rag-contract-management-stackit-aleph-alpha/) | Build a Region-Specific RAG System for Contract Management | Qdrant, Aleph Alpha, STACKIT |
|
||||
| [Question-Answering System for Customer Support](/documentation/examples/rag-customer-support-cohere-airbyte-aws/) | Build a RAG System for AI Customer Support | Qdrant, Cohere, Airbyte, AWS |
|
||||
| [Hybrid Search on PDF Documents](/documentation/examples/hybrid-search-llamaindex-jinaai/) | Develop a Hybrid Search System for Product PDF Manuals | Qdrant, LlamaIndex, Jina AI
|
||||
| [Blog-Reading RAG Chatbot](/documentation/examples/rag-chatbot-scaleway/) | Develop a RAG-based Chatbot on Scaleway and with LangChain | Qdrant, LangChain, GPT-4o
|
||||
| [Movie Recommendation System](/documentation/examples/recommendation-system-ovhcloud/) | Build a Movie Recommendation System with LlamaIndex and With JinaAI | Qdrant |
|
||||
|
||||
|
||||
## Notebooks
|
||||
@@ -28,7 +28,7 @@ Our Notebooks offer complex instructions that are supported with a throrough exp
|
||||
| [Search and Recommend Newspaper Articles](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_text_data/qdrant_and_text_data.ipynb) | Work with text data to develop a semantic search and a recommendation engine for news articles. | Qdrant |
|
||||
| [Recommendation System for Songs](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_audio_data/03_qdrant_101_audio.ipynb) | Use Qdrant to develop a music recommendation engine based on audio embeddings. | Qdrant |
|
||||
| [Image Comparison System for Skin Conditions](https://colab.research.google.com/github/qdrant/examples/blob/master/qdrant_101_image_data/04_qdrant_101_cv.ipynb) | Use Qdrant to compare challenging images with labels representing different skin diseases. | Qdrant |
|
||||
| [Question and Answer System with LlamaIndex](https://githubtocolab.com/qdrant/examples/blob/master/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb) | Combine Qdrant and LlamaIndex to create a self-updating Q&A system. | Qdrant, LlamaIndex, Cohere |
|
||||
| [Question and Answer System with LlamaIndex](https://github.com/qdrant/examples/blob/949669f001a03131afebf2ecd1e0ce63cab01c81/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb) | Combine Qdrant and LlamaIndex to create a self-updating Q&A system. | Qdrant, LlamaIndex, Cohere |
|
||||
| [Extractive QA System](https://githubtocolab.com/qdrant/examples/blob/master/extractive_qa/extractive-question-answering.ipynb) | Extract answers directly from context to generate highly relevant answers. | Qdrant |
|
||||
| [Ecommerce Reverse Image Search](https://githubtocolab.com/qdrant/examples/blob/master/ecommerce_reverse_image_search/ecommerce-reverse-image-search.ipynb) | Accept images as search queries to receive semantically appropriate answers. | Qdrant |
|
||||
| [Basic RAG](https://githubtocolab.com/qdrant/examples/blob/master/rag-openai-qdrant/rag-openai-qdrant.ipynb) | Basic RAG pipeline with Qdrant and OpenAI SDKs. | OpenAI, Qdrant, FastEmbed |
|
||||
|
||||
@@ -63,7 +63,7 @@ Verify that mighty works by calling `curl https://<address>:5050/sentence-transf
|
||||
}
|
||||
```
|
||||
|
||||
For Qdrant, follow our [cloud documentation](../../cloud/cloud-quick-start/) to spin up a [free tier](https://cloud.qdrant.io/). Make sure to retrieve an API key.
|
||||
For Qdrant, follow our [cloud documentation](/documentation/cloud/cloud-quick-start/) to spin up a [free tier](https://cloud.qdrant.io/). Make sure to retrieve an API key.
|
||||
|
||||
## Implement model API
|
||||
|
||||
|
||||
+1
-1
@@ -234,7 +234,7 @@ llm = AlephAlpha(
|
||||
Then, we can glue the components together and build the search process. `RetrievalQA` is a class that takes implements
|
||||
the Question Retrieval process, with a specified retriever and Large Language Model. The instance of `Qdrant` might be
|
||||
converted into a retriever, with additional filter that will be passed to the `similarity_search` method. The filter
|
||||
is created as [in a regular Qdrant query](../../../documentation/concepts/filtering/), with the `roles` field set to the
|
||||
is created as [in a regular Qdrant query](/documentation/concepts/filtering/), with the `roles` field set to the
|
||||
user's roles.
|
||||
|
||||
```python
|
||||
|
||||
+2
-2
@@ -132,14 +132,14 @@ progress of the synchronization in the UI.
|
||||
## RAG connector
|
||||
|
||||
One of our previous tutorials, guides you step-by-step on [implementing custom connector for Cohere
|
||||
RAG](../cohere-rag-connector/) with Cohere Embed v3 and Qdrant. You can just point it to use your Hybrid Cloud
|
||||
RAG](documentation/examples/cohere-rag-connector/) with Cohere Embed v3 and Qdrant. You can just point it to use your Hybrid Cloud
|
||||
Qdrant instance running on AWS. Created connector might be deployed to Amazon Web Services in various ways, even in a
|
||||
[Serverless](https://aws.amazon.com/serverless/) manner using [AWS
|
||||
Lambda](https://aws.amazon.com/lambda/?c=ser&sec=srv).
|
||||
|
||||
In general, RAG connector has to expose a single endpoint that will accept POST requests with `query` parameter and
|
||||
return the matching documents as JSON document with a specific structure. Our FastAPI implementation created [in the
|
||||
related tutorial](../cohere-rag-connector/) is a perfect fit for this task. The only difference is that you
|
||||
related tutorial](documentation/examples/cohere-rag-connector/) is a perfect fit for this task. The only difference is that you
|
||||
should point it to the Cohere models and Qdrant running on AWS infrastructure.
|
||||
|
||||
> Our connector is a lightweight web service that exposes a single endpoint and glues the Cohere embedding model with
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
title: FAQ
|
||||
weight: 27
|
||||
weight: 28
|
||||
is_empty: true
|
||||
---
|
||||
@@ -9,18 +9,18 @@ weight: 2
|
||||
|
||||
The primary source of memory usage is vector data. There are several ways to address that:
|
||||
|
||||
- Configure [Quantization](../../guides/quantization/) to reduce the memory usage of vectors.
|
||||
- Configure [Quantization](/documentation/guides/quantization/) to reduce the memory usage of vectors.
|
||||
- Configure on-disk vector storage
|
||||
|
||||
The choice of the approach depends on your requirements.
|
||||
Read more about [configuring the optimal](../../tutorials/optimize/) use of Qdrant.
|
||||
Read more about [configuring the optimal](/documentation/tutorials/optimize/) use of Qdrant.
|
||||
|
||||
### How do you choose the machine configuration?
|
||||
|
||||
There are two main scenarios of Qdrant usage in terms of resource consumption:
|
||||
|
||||
- **Performance-optimized** -- when you need to serve vector search as fast (many) as possible. In this case, you need to have as much vector data in RAM as possible. Use our [calculator](https://cloud.qdrant.io/calculator) to estimate the required RAM.
|
||||
- **Storage-optimized** -- when you need to store many vectors and minimize costs by compromising some search speed. In this case, pay attention to the disk speed instead. More about it in the article about [Memory Consumption](../../../articles/memory-consumption/).
|
||||
- **Storage-optimized** -- when you need to store many vectors and minimize costs by compromising some search speed. In this case, pay attention to the disk speed instead. More about it in the article about [Memory Consumption](/articles/memory-consumption/).
|
||||
|
||||
### I configured on-disk vector storage, but memory usage is still high. Why?
|
||||
|
||||
@@ -38,6 +38,6 @@ If you want to limit the memory usage of the service, we recommend using [limits
|
||||
|
||||
There are several possible reasons for that:
|
||||
|
||||
- **Using filters without payload index** -- If you're performing a search with a filter but you don't have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](../../concepts/indexing/#payload-index).
|
||||
- **Usage of on-disk vector storage with slow disks** -- If you're using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](../../../articles/memory-consumption/).
|
||||
- **Using filters without payload index** -- If you're performing a search with a filter but you don't have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](/documentation/concepts/indexing/#payload-index).
|
||||
- **Usage of on-disk vector storage with slow disks** -- If you're using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](/articles/memory-consumption/).
|
||||
- **Large limit or non-optimal query parameters** -- A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues.
|
||||
@@ -53,7 +53,7 @@ If you're still seeing `"vector": null` in your results, it might be that the ve
|
||||
|
||||
### How can I search without a vector?
|
||||
|
||||
You are likely looking for the [scroll](../../concepts/points/#scroll-points) method. It allows you to retrieve the records based on filters or even iterate over all the records in the collection.
|
||||
You are likely looking for the [scroll](/documentation/concepts/points/#scroll-points) method. It allows you to retrieve the records based on filters or even iterate over all the records in the collection.
|
||||
|
||||
### Does Qdrant support a full-text search or a hybrid search?
|
||||
|
||||
@@ -64,10 +64,10 @@ What Qdrant can do:
|
||||
|
||||
- Search with full-text filters
|
||||
- Apply full-text filters to the vector search (i.e., perform vector search among the records with specific words or phrases)
|
||||
- Do prefix search and semantic [search-as-you-type](../../../articles/search-as-you-type/)
|
||||
- Do prefix search and semantic [search-as-you-type](/articles/search-as-you-type/)
|
||||
- Sparse vectors, as used in [SPLADE](https://github.com/naver/splade) or similar models
|
||||
- [Multi-vectors](../../concepts/vectors/#multivectors), for example ColBERT and other late-interaction models
|
||||
- Combination of the [multiple searches](../../concepts/hybrid-queries/)
|
||||
- [Multi-vectors](/documentation/concepts/vectors/#multivectors), for example ColBERT and other late-interaction models
|
||||
- Combination of the [multiple searches](/documentation/concepts/hybrid-queries/)
|
||||
|
||||
What Qdrant doesn't plan to support:
|
||||
|
||||
@@ -76,7 +76,7 @@ What Qdrant doesn't plan to support:
|
||||
- Query analyzers and other NLP tools
|
||||
|
||||
Of course, you can always combine Qdrant with any specialized tool you need, including full-text search engines.
|
||||
Read more about [our approach](../../../articles/hybrid-search/) to hybrid search.
|
||||
Read more about [our approach](/articles/hybrid-search/) to hybrid search.
|
||||
|
||||
## Collections
|
||||
|
||||
@@ -87,11 +87,11 @@ It is _highly_ recommended not to create many small collections, as it will lead
|
||||
|
||||
We consider creating a collection for each user/dialog/document as an antipattern.
|
||||
|
||||
Please read more about collections, isolation, and multiple users in our [Multitenancy](../../tutorials/multiple-partitions/) tutorial.
|
||||
Please read more about collections, isolation, and multiple users in our [Multitenancy](/documentation/tutorials/multiple-partitions/) tutorial.
|
||||
|
||||
### How do I upload a large number of vectors into a Qdrant collection?
|
||||
|
||||
Read about our recommendations in the [bulk upload](../../tutorials/bulk-upload/) tutorial.
|
||||
Read about our recommendations in the [bulk upload](/documentation/tutorials/bulk-upload/) tutorial.
|
||||
|
||||
### Can I only store quantized vectors and discard full precision vectors?
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ FastEmbed easily integrates with Qdrant for a variety of multimodal search purpo
|
||||
|
||||
|Beginner|Advanced|
|
||||
|:-:|:-:|
|
||||
|[Generate Text Embedings with FastEmbed](fastembed-quickstart/)|[Combine FastEmbed with Qdrant for Vector Search](fastembed-semantic-search/)|
|
||||
|[Generate Text Embedings with FastEmbed](/documentation/fastembed/fastembed-quickstart/)|[Combine FastEmbed with Qdrant for Vector Search](/documentation/fastembed/fastembed-semantic-search/)|
|
||||
|
||||
## Why is FastEmbed useful?
|
||||
|
||||
|
||||
@@ -7,22 +7,22 @@ weight: 20
|
||||
|
||||
| Framework | Description |
|
||||
| ------------------------------------- | ---------------------------------------------------------------------------------------------------- |
|
||||
| [AutoGen](./autogen/) | Framework from Microsoft building LLM applications using multiple conversational agents. |
|
||||
| [Canopy](./canopy/) | Framework from Pinecone for building RAG applications using LLMs and knowledge bases. |
|
||||
| [Cheshire Cat](./cheshire-cat/) | Framework to create personalized AI assistants using custom data. |
|
||||
| [DocArray](./docarray/) | Python library for managing data in multi-modal AI applications. |
|
||||
| [DSPy](./dspy/) | Framework for algorithmically optimizing LM prompts and weights. |
|
||||
| [Fifty-One](./fifty-one/) | Toolkit for building high-quality datasets and computer vision models. |
|
||||
| [Genkit](./genkit/) | Framework to build, deploy, and monitor production-ready AI-powered apps. |
|
||||
| [Haystack](./haystack/) | LLM orchestration framework to build customizable, production-ready LLM applications. |
|
||||
| [Langchain](./langchain/) | Python framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [Langchain-Go](./langchain-go/) | Go framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [Langchain4j](./langchain4j/) | Java framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [LlamaIndex](./llama-index/) | A data framework for building LLM applications with modular integrations. |
|
||||
| [MemGPT](./memgpt/) | System to build LLM agents with long term memory & custom tools |
|
||||
| [Pandas-AI](./pandas-ai/) | Python library to query/visualize your data (CSV, XLSX, PostgreSQL, etc.) in natural language |
|
||||
| [Semantic Router](./semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. |
|
||||
| [Spring AI](./spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. |
|
||||
| [Testcontainers](./testcontainers/) | Set of frameworks for running containerized dependencies in tests. |
|
||||
| [txtai](./txtai/) | Python library for semantic search, LLM orchestration and language model workflows. |
|
||||
| [Vanna AI](./vanna-ai/) | Python RAG framework for SQL generation and querying. |
|
||||
| [AutoGen](/documentation/frameworks/autogen/) | Framework from Microsoft building LLM applications using multiple conversational agents. |
|
||||
| [Canopy](/documentation/frameworks/canopy/) | Framework from Pinecone for building RAG applications using LLMs and knowledge bases. |
|
||||
| [Cheshire Cat](/documentation/frameworks/cheshire-cat/) | Framework to create personalized AI assistants using custom data. |
|
||||
| [DocArray](/documentation/frameworks/docarray/) | Python library for managing data in multi-modal AI applications. |
|
||||
| [DSPy](/documentation/frameworks/dspy/) | Framework for algorithmically optimizing LM prompts and weights. |
|
||||
| [Fifty-One](/documentation/frameworks/fifty-one/) | Toolkit for building high-quality datasets and computer vision models. |
|
||||
| [Genkit](/documentation/frameworks/genkit/) | Framework to build, deploy, and monitor production-ready AI-powered apps. |
|
||||
| [Haystack](/documentation/frameworks/haystack/) | LLM orchestration framework to build customizable, production-ready LLM applications. |
|
||||
| [Langchain](/documentation/frameworks/langchain/) | Python framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [Langchain-Go](/documentation/frameworks/langchain-go/) | Go framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [Langchain4j](/documentation/frameworks/langchain4j/) | Java framework for building context-aware, reasoning applications using LLMs. |
|
||||
| [LlamaIndex](/documentation/frameworks/llama-index/) | A data framework for building LLM applications with modular integrations. |
|
||||
| [Mem0](/documentation/frameworks/mem0/) | Self-improving memory layer for LLM applications, enabling personalized AI experiences. |
|
||||
| [MemGPT](/documentation/frameworks/memgpt/) | System to build LLM agents with long term memory & custom tools |
|
||||
| [Pandas-AI](/documentation/frameworks/pandas-ai/) | Python library to query/visualize your data (CSV, XLSX, PostgreSQL, etc.) in natural language |
|
||||
| [Semantic Router](/documentation/frameworks/semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. |
|
||||
| [Spring AI](/documentation/frameworks/spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. |
|
||||
| [txtai](/documentation/frameworks/txtai/) | Python library for semantic search, LLM orchestration and language model workflows. |
|
||||
| [Vanna AI](/documentation/frameworks/vanna-ai/) | Python RAG framework for SQL generation and querying. |
|
||||
|
||||
@@ -25,9 +25,9 @@ CORE_PORT=1865
|
||||
|
||||
Cheshire Cat takes great advantage of the following features of Qdrant:
|
||||
|
||||
* [Collection Aliases](../../concepts/collections/#collection-aliases) to manage the change from one embedder to another.
|
||||
* [Quantization](../../guides/quantization/) to obtain a good balance between speed, memory usage and quality of the results.
|
||||
* [Snapshots](../../concepts/snapshots/) to not miss any information.
|
||||
* [Collection Aliases](/documentation/concepts/collections/#collection-aliases) to manage the change from one embedder to another.
|
||||
* [Quantization](/documentation/guides/quantization/) to obtain a good balance between speed, memory usage and quality of the results.
|
||||
* [Snapshots](/documentation/concepts/snapshots/) to not miss any information.
|
||||
* [Community](https://discord.com/invite/tdtYvXjC4h)
|
||||
|
||||

|
||||
|
||||
@@ -64,7 +64,7 @@ addition, there are a few optional parameters:
|
||||
metadataPayloadKey: 'metadata';
|
||||
```
|
||||
|
||||
- `collectionCreateOptions`: [Additional options](<(https://qdrant.tech/documentation/concepts/collections/#create-a-collection)>) when creating the Qdrant collection.
|
||||
- `collectionCreateOptions`: [Additional options](/documentation/concepts/collections/#create-a-collection/) when creating the Qdrant collection.
|
||||
|
||||
## Usage
|
||||
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
title: Mem0
|
||||
---
|
||||
|
||||

|
||||
|
||||
[Mem0](https://mem0.ai) is a self-improving memory layer for LLM applications, enabling personalized AI experiences that save costs and delight users. Mem0 remembers user preferences, adapts to individual needs, and continuously improves over time, ideal for chatbots and AI systems.
|
||||
|
||||
Mem0 supports various vector store providers, including Qdrant, for efficient data handling and search capabilities.
|
||||
|
||||
## Installation
|
||||
|
||||
To install Mem0 with Qdrant support, use the following command:
|
||||
|
||||
```sh
|
||||
pip install mem0ai
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Here's a basic example of how to use Mem0 with Qdrant:
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-xx"
|
||||
|
||||
config = {
|
||||
"vector_store": {
|
||||
"provider": "qdrant",
|
||||
"config": {
|
||||
"collection_name": "test",
|
||||
"host": "localhost",
|
||||
"port": 6333,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
When configuring Mem0 to use Qdrant as the vector store, you can specify [various parameters](https://docs.mem0.ai/components/vectordbs/dbs/qdrant#config) in the `config` dictionary.
|
||||
|
||||
## Advanced Usage
|
||||
|
||||
Mem0 provides additional functionality for managing and querying your vector data. Here are some examples:
|
||||
|
||||
```python
|
||||
# Search memories
|
||||
related_memories = m.search(query="What are Alice's hobbies?", user_id="alice")
|
||||
|
||||
# Update existing memory
|
||||
result = m.update(memory_id="m1", data="Likes to play tennis on weekends")
|
||||
|
||||
# Get memory history
|
||||
history = m.history(memory_id="m1")
|
||||
```
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Mem0 GitHub Repository](https://github.com/mem0ai/mem0)
|
||||
- [Mem0 Documentation](https://docs.mem0.ai/)
|
||||
@@ -44,7 +44,7 @@ example, by deleting a collection. After resolving Qdrant can be restarted
|
||||
normally to continue operation.
|
||||
|
||||
In recovery mode, collection operations are limited to
|
||||
[deleting](../../concepts/collections/#delete-collection) a
|
||||
[deleting](/documentation/concepts/collections/#delete-collection) a
|
||||
collection. That is because only collection metadata is loaded during recovery.
|
||||
|
||||
To enable recovery mode with the Qdrant Docker image you must set the
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
---
|
||||
title: Capacity Planning
|
||||
weight: 11
|
||||
aliases:
|
||||
- capacity
|
||||
- /documentation/cloud/capacity-sizing
|
||||
---
|
||||
# Capacity Planning
|
||||
|
||||
When setting up your cluster, you'll need to figure out the right balance of **RAM** and **disk storage**. The best setup depends on a few things:
|
||||
|
||||
- How many vectors you have and their dimensions.
|
||||
- The amount of payload data you're using and their indexes.
|
||||
- What data you want to store in memory versus on disk.
|
||||
- Your cluster's replication settings.
|
||||
- Whether you're using quantization and how you’ve set it up.
|
||||
|
||||
## Calculating RAM size
|
||||
|
||||
You should store frequently accessed data in RAM for faster retrieval. If you want to keep all vectors in memory for optimal performance, you can use this rough formula for estimation:
|
||||
|
||||
```text
|
||||
memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5
|
||||
```
|
||||
|
||||
At the end, we multiply everything by 1.5. This extra 50% accounts for metadata (such as indexes and point versions) and temporary segments created during optimization.
|
||||
|
||||
Let's say you want to store 1 million vectors with 1024 dimensions:
|
||||
|
||||
```text
|
||||
memory_size = 1,000,000 * 1024 * 4 bytes * 1.5
|
||||
```
|
||||
The memory_size is approximately 6,144,000,000 bytes, or about 5.72 GB.
|
||||
|
||||
Depending on the use case, large datasets can benefit from reduced memory requirements via [quantization](/documentation/guides/quantization/).
|
||||
|
||||
## Calculating payload size
|
||||
|
||||
This is always different. The size of the payload depends on the [structure and content of your data](/documentation/concepts/payload/#payload-types). For instance:
|
||||
|
||||
- **Text fields** consume space based on length and encoding (e.g. a large chunk of text vs a few words).
|
||||
- **Floats** have fixed sizes of 8 bytes for `int64` or `float64`.
|
||||
- **Boolean fields** typically consume 1 byte.
|
||||
|
||||
<aside role="alert">
|
||||
The easiest way to calculate your payload size is to use a JSON size calculator.
|
||||
</aside>
|
||||
|
||||
Calculating total payload size is similar to vectors. We have to multiply it by 1.5 for back-end indexing processes.
|
||||
|
||||
```text
|
||||
total_payload_size = number_of_points * payload_size * 1.5
|
||||
```
|
||||
|
||||
Let's say you want to store 1 million points with JSON payloads of 5KB:
|
||||
|
||||
```text
|
||||
total_payload_size = 1,000,000 * 5KB * 1.5
|
||||
```
|
||||
The total_payload_size is approximately 5,000,000 bytes, or about 4.77 GB.
|
||||
|
||||
## Choosing disk over RAM
|
||||
|
||||
For optimal performance, you should store only frequently accessed data in RAM. The rest should be offloaded to the disk. For example, extra payload fields that you don't use for filtering can be stored on disk.
|
||||
|
||||
Only [indexed fields](/documentation/concepts/indexing/#payload-index) should be stored in RAM. You can read more about payload storage in the [Storage](/documentation/concepts/storage/#payload-storage) section.
|
||||
|
||||
### Storage-focused configuration
|
||||
|
||||
If your priority is to handle large volumes of vectors with average search latency, it's recommended to configure [memory-mapped (mmap) storage](/documentation/concepts/storage/#configuring-memmap-storage). In this setup, vectors are stored on disk in memory-mapped files, while only the most frequently accessed vectors are cached in RAM.
|
||||
|
||||
The amount of available RAM greatly impacts search performance. As a general rule, if you store half as many vectors in RAM, search latency will roughly double.
|
||||
|
||||
Disk speed is also crucial. [Contact us](/documentation/support/) if you have specific requirements for high-volume searches in our Cloud.
|
||||
|
||||
### Subgroup-oriented configuration
|
||||
|
||||
If your use case involves splitting vectors into multiple collections or subgroups based on payload values (e.g., serving searches for multiple users, each with their own subset of vectors), memory-mapped storage is recommended.
|
||||
|
||||
In this scenario, only the active subset of vectors will be cached in RAM, allowing for fast searches for the most recent and active users. You can estimate the required memory size as:
|
||||
|
||||
```text
|
||||
memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5
|
||||
```
|
||||
|
||||
Please refer to our [multitenancy](/documentation/guides/multiple-partitions/) documentation for more details on partitioning data in a Qdrant.
|
||||
|
||||
## Scaling disk space in Qdrant Cloud
|
||||
|
||||
Clusters supporting vector search require substantial disk space compared to other search systems. If you're running low on disk space, you can use the UI at [cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster.
|
||||
|
||||
<aside role="status">Note: If you increase disk space via the Qdrant UI, you cannot reduce it later.</aside>
|
||||
|
||||
When running low on disk space, consider the following benefits of scaling up:
|
||||
|
||||
- **Larger Datasets**: Supports larger datasets, which can improve the relevance and quality of search results.
|
||||
- **Improved Indexing**: Enables the use of advanced indexing strategies like HNSW.
|
||||
- **Caching**: Enhances speed by having more RAM, allowing more frequently accessed data to be cached.
|
||||
- **Backups and Redundancy**: Facilitates more frequent backups, which is a key advantage for data safety.
|
||||
|
||||
Always remember to add 50% of the vector size. This would account for things like indexes and auxiliary data used during operations such as vector insertion, deletion, and search. Thus, the estimated memory size including metadata is:
|
||||
|
||||
```text
|
||||
total_vector_size = number_of_dimensions * 4 bytes * 1.5
|
||||
```
|
||||
|
||||
**Disclaimer**
|
||||
|
||||
The above calculations are estimates at best. If you're looking for more accurate numbers, you should always test your data set in practice.
|
||||
@@ -42,7 +42,7 @@ Can't open Collections meta Wal: Os { code: 11, kind: WouldBlock, message: "Reso
|
||||
```
|
||||
|
||||
It means that Qdrant cannot start because a collection cannot be loaded. Its
|
||||
associated [WAL](../../concepts/storage/#versioning) files are currently
|
||||
associated [WAL](/documentation/concepts/storage/#versioning) files are currently
|
||||
unavailable, likely because the same files are already being used by another
|
||||
Qdrant instance.
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ production mode, you could also choose to overwrite `config/production.yaml`.
|
||||
See [ordering](#order-and-priority) for details on how configurations are
|
||||
loaded.
|
||||
|
||||
The [Installation](../installation/) guide contains examples of how to set up Qdrant with a custom configuration for the different deployment methods.
|
||||
The [Installation](/documentation/guides/installation/) guide contains examples of how to set up Qdrant with a custom configuration for the different deployment methods.
|
||||
|
||||
## Order and priority
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ In summary, single-node clusters are best for non-production workloads, replicat
|
||||
|
||||
## Enabling distributed mode in self-hosted Qdrant
|
||||
|
||||
To enable distributed deployment - enable the cluster mode in the [configuration](../configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`.
|
||||
To enable distributed deployment - enable the cluster mode in the [configuration](/documentation/guides/configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`.
|
||||
|
||||
```yaml
|
||||
cluster:
|
||||
@@ -152,7 +152,7 @@ Qdrant uses the [Raft](https://raft.github.io/) consensus protocol to maintain c
|
||||
|
||||
Operations on points, on the other hand, do not go through the consensus infrastructure.
|
||||
Qdrant is not intended to have strong transaction guarantees, which allows it to perform point operations with low overhead.
|
||||
In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](../../concepts/points/#awaiting-result) to see the results of your writes.
|
||||
In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](/documentation/concepts/points/#awaiting-result) to see the results of your writes.
|
||||
|
||||
Operations on collections, on the contrary, are part of the consensus which guarantees that all operations are durable and eventually executed by all nodes.
|
||||
In practice it means that a majority of nodes agree on what operations should be applied before the service will perform them.
|
||||
@@ -171,7 +171,7 @@ There are two methods of distributing points across shards:
|
||||
|
||||
- **User-defined sharding**: _Available as of v1.7.0_ - Each point is uploaded to a specific shard, so that operations can hit only the shard or shards they need. Even with this distribution, shards still ensure having non-intersecting subsets of points. [See more...](#user-defined-sharding)
|
||||
|
||||
Each node knows where all parts of the collection are stored through the [consensus protocol](./#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result.
|
||||
Each node knows where all parts of the collection are stored through the [consensus protocol](#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result.
|
||||
|
||||
### Choosing the right number of shards
|
||||
|
||||
@@ -667,7 +667,7 @@ fastest depends on the size and state of a shard.
|
||||
Available shard transfer methods are:
|
||||
|
||||
- `stream_records`: _(default)_ transfer by streaming just its records to the target node in batches.
|
||||
- `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](../../concepts/snapshots/) automatically.
|
||||
- `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](/documentation/concepts/snapshots/) automatically.
|
||||
- `wal_delta`: _(auto recovery default)_ transfer by resolving [WAL] difference; the operations that were missed.
|
||||
|
||||
Each has pros, cons and specific requirements, some of which are:
|
||||
@@ -720,7 +720,7 @@ are acceptable in your use case. If your cluster is unstable and out of
|
||||
resources, it's probably best to use the `stream_records` transfer method,
|
||||
because it is unlikely to fail.
|
||||
|
||||
The `snapshot` transfer method utilizes [snapshots](../../concepts/snapshots/)
|
||||
The `snapshot` transfer method utilizes [snapshots](/documentation/concepts/snapshots/)
|
||||
to transfer a shard. A snapshot is created automatically. It is then transferred
|
||||
and restored on the target node. After this is done, the snapshot is removed
|
||||
from both nodes. While the snapshot/transfer/restore operation is happening, the
|
||||
@@ -749,7 +749,7 @@ The `stream_records` method is currently used as default. This may change in the
|
||||
future. As of Qdrant 1.9.0 `wal_delta` is used for automatic shard replications
|
||||
to recover dead shards.
|
||||
|
||||
[WAL]: ../../concepts/storage/#versioning
|
||||
[WAL]: /documentation/concepts/storage/#versioning
|
||||
|
||||
## Replication
|
||||
|
||||
@@ -985,7 +985,7 @@ Snapshot recovery, used in single-node deployment, is different from cluster one
|
||||
Consensus manages all metadata about all collections and does not require snapshots to recover it.
|
||||
But you can use snapshots to recover missing shards of the collections.
|
||||
|
||||
Use the [Collection Snapshot Recovery API](../../concepts/snapshots/#recover-in-cluster-deployment) to do it.
|
||||
Use the [Collection Snapshot Recovery API](/documentation/concepts/snapshots/#recover-in-cluster-deployment) to do it.
|
||||
The service will download the specified snapshot of the collection and recover shards with data from it.
|
||||
|
||||
Once all shards of the collection are recovered, the collection will become operational again.
|
||||
|
||||
@@ -6,13 +6,13 @@ aliases:
|
||||
- ../installation
|
||||
---
|
||||
|
||||
## Installation requirements
|
||||
# Installation requirements
|
||||
|
||||
The following sections describe the requirements for deploying Qdrant.
|
||||
|
||||
### CPU and memory
|
||||
## CPU and memory
|
||||
|
||||
The CPU and RAM that you need depends on:
|
||||
The preferred size of your CPU and RAM depends on:
|
||||
|
||||
- Number of vectors
|
||||
- Vector dimensions
|
||||
@@ -23,6 +23,15 @@ The CPU and RAM that you need depends on:
|
||||
|
||||
Our [Cloud Pricing Calculator](https://cloud.qdrant.io/calculator) can help you estimate required resources without payload or index data.
|
||||
|
||||
### Supported CPU architectures:
|
||||
|
||||
**64-bit system:**
|
||||
- x86_64/amd64
|
||||
- AArch64/arm64
|
||||
|
||||
**32-bit system:**
|
||||
- Not supported
|
||||
|
||||
### Storage
|
||||
|
||||
For persistent storage, Qdrant requires block-level access to storage devices with a [POSIX-compatible file system](https://www.quobyte.com/storage-explained/posix-filesystem/). Network systems such as [iSCSI](https://en.wikipedia.org/wiki/ISCSI) that provide block-level access are also acceptable.
|
||||
@@ -197,4 +206,4 @@ After a successful build, you can find the binary in the following subdirectory
|
||||
|
||||
## Client libraries
|
||||
|
||||
In addition to the service, Qdrant provides a variety of client libraries for different programming languages. For a full list, see our [Client libraries](../../interfaces/#client-libraries) documentation.
|
||||
In addition to the service, Qdrant provides a variety of client libraries for different programming languages. For a full list, see our [Client libraries](/documentation/interfaces/#client-libraries) documentation.
|
||||
|
||||
@@ -75,7 +75,7 @@ Qdrant server.
|
||||
These currently provide the most basic status response, returning HTTP 200 if
|
||||
Qdrant is started and ready to be used.
|
||||
|
||||
Regardless of whether an [API key](../security/#authentication) is configured,
|
||||
Regardless of whether an [API key](/documentation/guides/security/#authentication) is configured,
|
||||
the endpoints are always accessible.
|
||||
|
||||
You can read more about Kubernetes health endpoints
|
||||
|
||||
@@ -1,26 +1,30 @@
|
||||
---
|
||||
title: Optimize Resources
|
||||
title: Optimize Performance
|
||||
weight: 11
|
||||
aliases:
|
||||
- ../tutorials/optimize
|
||||
---
|
||||
|
||||
# Optimize Qdrant
|
||||
# Optimizing Qdrant Performance: Three Scenarios
|
||||
|
||||
Different use cases have different requirements for balancing between memory, speed, and precision.
|
||||
Qdrant is designed to be flexible and customizable so you can tune it to your needs.
|
||||
Different use cases require different balances between memory usage, search speed, and precision. Qdrant is designed to be flexible and customizable so you can tune it to your specific needs.
|
||||
|
||||

|
||||
This guide will walk you three main optimization strategies:
|
||||
- High Speed Search & Low Memory Usage
|
||||
- High Precision & Low Memory Usage
|
||||
- High Precision & High Speed Search
|
||||
|
||||
Let's look deeper into each of those possible optimization scenarios.
|
||||

|
||||
|
||||
## Prefer low memory footprint with high speed search
|
||||
## 1. High-Speed Search with Low Memory Usage
|
||||
|
||||
The main way to achieve high speed search with low memory footprint is to keep vectors on disk while at the same time minimizing the number of disk reads.
|
||||
To achieve high search speed with minimal memory usage, you can store vectors on disk while minimizing the number of disk reads. Vector quantization is a technique that compresses vectors, allowing more of them to be stored in memory, thus reducing the need to read from disk.
|
||||
|
||||
Vector quantization is one way to achieve this. Quantization converts vectors into a more compact representation, which can be stored in memory and used for search. With smaller vectors you can cache more in RAM and reduce the number of disk reads.
|
||||
To configure in-memory quantization, with on-disk original vectors, you need to create a collection with the following parameters:
|
||||
|
||||
To configure in-memory quantization, with on-disk original vectors, you need to create a collection with the following configuration:
|
||||
- `on_disk`: Stores original vectors on disk.
|
||||
- `quantization_config`: Compresses quantized vectors to `int8` using the `scalar` method.
|
||||
- `always_ram`: Keeps quantized vectors in RAM.
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -180,9 +184,9 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
})
|
||||
```
|
||||
|
||||
`on_disk` will ensure that vectors will be stored on disk, while `always_ram` will ensure that quantized vectors will be stored in RAM.
|
||||
### Disable Rescoring for Faster Search (optional)
|
||||
|
||||
Optionally, you can disable rescoring with search `params`, which will reduce the number of disk reads even further, but potentially slightly decrease the precision.
|
||||
This is completely optional. Disabling rescoring with search `params` can further reduce the number of disk reads. Note that this might slightly decrease precision.
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
@@ -313,9 +317,11 @@ client.Query(context.Background(), &qdrant.QueryPoints{
|
||||
})
|
||||
```
|
||||
|
||||
## Prefer high precision with low memory footprint
|
||||
## 2. High Precision with Low Memory Usage
|
||||
|
||||
In case you need high precision, but don't have enough RAM to store vectors in memory, you can enable on-disk vectors and HNSW index.
|
||||
If you require high precision but have limited RAM, you can store both vectors and the HNSW index on disk. This setup reduces memory usage while maintaining search precision.
|
||||
|
||||
To store the vectors `on_disk`, you need to configure both the vectors and the HNSW index:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -446,7 +452,9 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
})
|
||||
```
|
||||
|
||||
In this scenario you can increase the precision of the search by increasing the `ef` and `m` parameters of the HNSW index, even with limited RAM.
|
||||
### Improving Precision
|
||||
|
||||
Increase the `ef` and `m` parameters of the HNSW index to improve precision, even with limited RAM:
|
||||
|
||||
```json
|
||||
...
|
||||
@@ -458,15 +466,14 @@ In this scenario you can increase the precision of the search by increasing the
|
||||
...
|
||||
```
|
||||
|
||||
The disk IOPS is a critical factor in this scenario, it will determine how fast you can perform search.
|
||||
**Note:** The speed of this setup depends on the disk’s IOPS (Input/Output Operations Per Second).</br>
|
||||
You can use [fio](https://gist.github.com/superboum/aaa45d305700a7873a8ebbab1abddf2b) to measure disk IOPS.
|
||||
|
||||
## Prefer high precision with high speed search
|
||||
## 3. High Precision with High-Speed Search
|
||||
|
||||
For high speed and high precision search it is critical to keep as much data in RAM as possible.
|
||||
By default, Qdrant follows this approach, but you can tune it to your needs.
|
||||
For scenarios requiring both high speed and high precision, keep as much data in RAM as possible. Apply quantization with re-scoring for tunable accuracy.
|
||||
|
||||
It is possible to achieve high search speed and tunable accuracy by applying quantization with re-scoring.
|
||||
Here is how you can configure scalar quantization for a collection:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -622,7 +629,13 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
})
|
||||
```
|
||||
|
||||
There are also some search-time parameters you can use to tune the search accuracy and speed:
|
||||
### Fine-Tuning Search Parameters
|
||||
|
||||
You can adjust search parameters like `hnsw_ef` and `exact` to balance between speed and precision:
|
||||
|
||||
**Key Parameters:**
|
||||
- `hnsw_ef`: Number of neighbors to visit during search (higher value = better accuracy, slower speed).
|
||||
- `exact`: Set to `true` for exact search, which is slower but more accurate. You can use it to compare results of the search with different `hnsw_ef` values versus the ground truth.
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
@@ -737,19 +750,20 @@ client.Query(context.Background(), &qdrant.QueryPoints{
|
||||
})
|
||||
```
|
||||
|
||||
- `hnsw_ef` - controls the number of neighbors to visit during search. The higher the value, the more accurate and slower the search will be. Recommended range is 32-512.
|
||||
- `exact` - if set to `true`, will perform exact search, which will be slower, but more accurate. You can use it to compare results of the search with different `hnsw_ef` values versus the ground truth.
|
||||
## Balancing Latency and Throughput
|
||||
|
||||
## Latency vs Throughput
|
||||
When optimizing search performance, latency and throughput are two main metrics to consider:
|
||||
- **Latency:** Time taken for a single request.
|
||||
- **Throughput:** Number of requests handled per second.
|
||||
|
||||
- There are two main approaches to measure the speed of search:
|
||||
- latency of the request - the time from the moment request is submitted to the moment a response is received
|
||||
- throughput - the number of requests per second the system can handle
|
||||
The following optimization approaches are not mutually exclusive, but in some cases it might be preferable to optimize for one or another.
|
||||
|
||||
Those approaches are not mutually exclusive, but in some cases it might be preferable to optimize for one or another.
|
||||
### Minimizing Latency
|
||||
|
||||
To prefer minimizing latency, you can set up Qdrant to use as many cores as possible for a single request\.
|
||||
You can do this by setting the number of segments in the collection to be equal to the number of cores in the system. In this case, each segment will be processed in parallel, and the final result will be obtained faster.
|
||||
To minimize latency, you can set up Qdrant to use as many cores as possible for a single request.
|
||||
You can do this by setting the number of segments in the collection to be equal to the number of cores in the system.
|
||||
|
||||
In this case, each segment will be processed in parallel, and the final result will be obtained faster.
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -877,10 +891,13 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
},
|
||||
})
|
||||
```
|
||||
### Maximizing Throughput
|
||||
|
||||
To prefer throughput, you can set up Qdrant to use as many cores as possible for processing multiple requests in parallel.
|
||||
To do that, you can configure qdrant to use minimal number of segments, which is usually 2.
|
||||
Large segments benefit from the size of the index and overall smaller number of vector comparisons required to find the nearest neighbors. But at the same time require more time to build index.
|
||||
To maximize throughput, configure Qdrant to use as many cores as possible to process multiple requests in parallel.
|
||||
|
||||
To do that, use fewer segments (usually 2) to handle more requests in parallel.
|
||||
|
||||
Large segments benefit from the size of the index and overall smaller number of vector comparisons required to find the nearest neighbors. However, they will require more time to build the HNSW index.
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
@@ -1008,3 +1025,13 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
|
||||
},
|
||||
})
|
||||
```
|
||||
|
||||
## Summary
|
||||
|
||||
By adjusting configurations like vector storage, quantization, and search parameters, you can optimize Qdrant for different use cases:
|
||||
- **Low Memory + High Speed:** Use vector quantization.
|
||||
- **High Precision + Low Memory:** Store vectors and HNSW index on disk.
|
||||
- **High Precision + High Speed:** Keep data in RAM, use quantization with re-scoring.
|
||||
- **Latency vs. Throughput:** Adjust segment numbers based on the priority.
|
||||
|
||||
Choose the strategy that best fits your use case to get the most out of Qdrant’s performance capabilities.
|
||||
@@ -0,0 +1,12 @@
|
||||
---
|
||||
title: Infrastructure
|
||||
weight: 21
|
||||
---
|
||||
|
||||
## Infrastructure Integrations
|
||||
|
||||
| Integration | Description |
|
||||
| ----------------------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| [Pulumi](/documentation/infrastructure/pulumi/) | Infrastructure as code tool for creating, deploying, and managing cloud infrastructure |
|
||||
| [Terraform](/documentation/infrastructure/terraform/) | infrastructure as code tool to define resources in human-readable configuration files. |
|
||||
| [Testcontainers](/documentation/infrastructure/testcontainers/) | Open source framework for providing throwaway, lightweight instances of systems for testing |
|
||||
@@ -0,0 +1,225 @@
|
||||
---
|
||||
title: Pulumi
|
||||
aliases: [ ../platforms/pulumi ]
|
||||
---
|
||||
|
||||

|
||||
|
||||
Pulumi is an open source infrastructure as code tool for creating, deploying, and managing cloud infrastructure.
|
||||
|
||||
A Qdrant SDK in any of Pulumi's supported languages can be generated based on the [Qdrant Terraform Provider](https://registry.terraform.io/providers/qdrant/qdrant-cloud/latest).
|
||||
|
||||
## Pre-requisites
|
||||
|
||||
1. A [Pulumi Installation](https://www.pulumi.com/docs/install/).
|
||||
2. An [API key](/documentation/qdrant-cloud-api/#authentication-connecting-to-cloud-api) to access the Qdrant cloud API.
|
||||
|
||||
## Setup
|
||||
|
||||
- Create a Pulumi project in any of the [supported languages](https://www.pulumi.com/docs/languages-sdks/) by running
|
||||
|
||||
```bash
|
||||
mkdir qdrant-pulumi && cd qdrant-pulumi
|
||||
pulumi new "<LANGUAGE>" -y
|
||||
```
|
||||
|
||||
- Generate a Pulumi SDK for Qdrant by running the following in your Pulumi project directory.
|
||||
|
||||
```bash
|
||||
pulumi package add terraform-provider registry.terraform.io/qdrant/qdrant-cloud
|
||||
```
|
||||
|
||||
- Set the Qdrant cloud API as a config value.
|
||||
|
||||
```bash
|
||||
pulumi config set qdrant-cloud:apiKey "<QDRANT_CLOUD_API_KEY>" --secret
|
||||
```
|
||||
|
||||
- You can now import the SDK as:
|
||||
|
||||
```python
|
||||
import pulumi_qdrant_cloud as qdrant_cloud
|
||||
```
|
||||
|
||||
```typescript
|
||||
import * as qdrantCloud from "qdrant-cloud";
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.*;
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
The provider includes the following data-sources and resources to work with:
|
||||
|
||||
### Data Sources
|
||||
|
||||
- `qdrant-cloud_booking_packages` - Get IDs and detailed information about the packages/subscriptions available. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/booking_packages.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.get_booking_packages(cloud_provider="aws", cloud_region="us-west-2")
|
||||
```
|
||||
|
||||
```typescript
|
||||
qdrantCloud.getBookingPackages({
|
||||
cloudProvider: "aws",
|
||||
cloudRegion: "us-west-2"
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.inputs.GetBookingPackagesArgs;
|
||||
|
||||
QdrantcloudFunctions.getBookingPackages(GetBookingPackagesArgs.builder()
|
||||
.cloudProvider("aws")
|
||||
.cloudRegion("us-west-2")
|
||||
.build());
|
||||
```
|
||||
|
||||
- `qdrant-cloud_accounts_auth_keys` - List API keys for Qdrant clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_auth_keys.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.get_accounts_auth_keys(account_id="<ACCOUNT_ID>")
|
||||
```
|
||||
|
||||
```typescript
|
||||
qdrantCloud.getAccountsAuthKeys({
|
||||
accountId: "<ACCOUNT_ID>"
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.inputs.GetAccountsAuthKeysArgs;
|
||||
|
||||
QdrantcloudFunctions.getAccountsAuthKeys(GetAccountsAuthKeysArgs.builder()
|
||||
.accountId("<ACCOUNT_ID>")
|
||||
.build());
|
||||
```
|
||||
|
||||
- `qdrant-cloud_accounts_cluster` - Get Cluster Information. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_cluster.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.get_accounts_cluster(
|
||||
account_id="<ACCOUNT_ID>",
|
||||
id="<CLUSTER_ID>",
|
||||
)
|
||||
```
|
||||
|
||||
```typescript
|
||||
qdrantCloud.getAccountsCluster({
|
||||
accountId: "<ACCOUNT_ID>",
|
||||
id: "<CLUSTER_ID>"
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.inputs.GetAccountsClusterArgs;
|
||||
|
||||
QdrantcloudFunctions.getAccountsCluster(GetAccountsClusterArgs
|
||||
.builder()
|
||||
.accountId("<ACCOUNT_ID>")
|
||||
.id("<CLUSTER_ID>")
|
||||
.build());
|
||||
```
|
||||
|
||||
- `qdrant-cloud_accounts_clusters` - List Qdrant clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_clusters.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.get_accounts_clusters(account_id="<ACCOUNT_ID>")
|
||||
```
|
||||
|
||||
```typescript
|
||||
qdrantCloud.getAccountsClusters({
|
||||
accountId: "<ACCOUNT_ID>"
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.inputs.GetAccountsClustersArgs;
|
||||
|
||||
QdrantcloudFunctions.getAccountsClusters(
|
||||
GetAccountsClustersArgs.builder().accountId("<ACCOUNT_ID>").build());
|
||||
```
|
||||
|
||||
### Resources
|
||||
|
||||
- `qdrant-cloud_accounts_cluster` - Create clusters on Qdrant cloud - [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/resources/accounts_cluster.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.AccountsCluster(
|
||||
resource_name="pl-example-cluster-resource",
|
||||
name="pl-example-cluster",
|
||||
cloud_provider="gcp",
|
||||
cloud_region="us-east4",
|
||||
configuration=qdrant_cloud.AccountsClusterConfigurationArgs(
|
||||
number_of_nodes=1,
|
||||
node_configuration=qdrant_cloud.AccountsClusterConfigurationNodeConfigurationArgs(
|
||||
package_id="3920d1eb-d3eb-4117-9578-b12d89bb1c5d"
|
||||
),
|
||||
),
|
||||
account_id="<ACCOUNT_ID>",
|
||||
)
|
||||
```
|
||||
|
||||
```typescript
|
||||
new qdrantCloud.AccountsCluster("pl-example-cluster-resource", {
|
||||
cloudProvider: "gcp",
|
||||
cloudRegion: "us-east4",
|
||||
configuration: {
|
||||
numberOfNodes: 1,
|
||||
nodeConfiguration: {
|
||||
packageId: "3920d1eb-d3eb-4117-9578-b12d89bb1c5d"
|
||||
}
|
||||
},
|
||||
accountId: "<ACCOUNT_ID>"
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.AccountsClusterArgs;
|
||||
import com.pulumi.qdrantcloud.inputs.AccountsClusterConfigurationArgs;
|
||||
import com.pulumi.qdrantcloud.inputs.AccountsClusterConfigurationNodeConfigurationArgs;
|
||||
|
||||
new AccountsCluster("pl-example-cluster-resource", AccountsClusterArgs.builder()
|
||||
.name("pl-example-cluster")
|
||||
.cloudProvider("gcp")
|
||||
.cloudRegion("us-east4")
|
||||
.configuration(AccountsClusterConfigurationArgs.builder()
|
||||
.numberOfNodes(1.0)
|
||||
.nodeConfiguration(AccountsClusterConfigurationNodeConfigurationArgs.builder()
|
||||
.packageId("3920d1eb-d3eb-4117-9578-b12d89bb1c5d")
|
||||
.build())
|
||||
.build())
|
||||
.accountId("<ACCOUNT_ID>")
|
||||
.build());
|
||||
```
|
||||
|
||||
- `qdrant-cloud_accounts_auth_key` - Create API keys for Qdrant cloud clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/resources/accounts_auth_key.md)
|
||||
|
||||
```python
|
||||
qdrant_cloud.AccountsAuthKey(
|
||||
resource_name="pl-example-key-resource",
|
||||
cluster_ids=["<CLUSTER_ID>"],
|
||||
)
|
||||
```
|
||||
|
||||
```typescript
|
||||
new qdrantCloud.AccountsAuthKey("pl-example-cluster-resource", {
|
||||
clusterIds: ["<CLUSTER_ID>", "<CLUSTER_ID_2>"]
|
||||
})
|
||||
```
|
||||
|
||||
```java
|
||||
import com.pulumi.qdrantcloud.AccountsAuthKey;
|
||||
import com.pulumi.qdrantcloud.AccountsAuthKeyArgs;
|
||||
|
||||
new AccountsAuthKey("pl-example-key-resource", AccountsAuthKeyArgs.builder()
|
||||
.clusterIds("<CLUSTER_ID>", "<CLUSTER_ID_2>")
|
||||
.build());
|
||||
```
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Provider Documentation](https://registry.terraform.io/providers/qdrant/qdrant-cloud/latest/docs)
|
||||
- [Pulumi Quickstart](https://www.pulumi.com/docs/get-started/)
|
||||
@@ -0,0 +1,77 @@
|
||||
---
|
||||
title: Terraform
|
||||
aliases: [ ../platforms/terraform ]
|
||||
---
|
||||
|
||||

|
||||
|
||||
HashiCorp Terraform is an infrastructure as code tool that lets you define both cloud and on-prem resources in human-readable configuration files that you can version, reuse, and share. You can then use a consistent workflow to provision and manage all of your infrastructure throughout its lifecycle.
|
||||
|
||||
With the [Qdrant Terraform Provider](https://registry.terraform.io/providers/qdrant/qdrant-cloud/latest), you can manage the Qdrant cloud lifecycle leveraging all the goodness of Terraform.
|
||||
|
||||
## Pre-requisites
|
||||
|
||||
To use the Qdrant Terraform Provider, you'll need:
|
||||
|
||||
1. A [Terraform installation](https://developer.hashicorp.com/terraform/install).
|
||||
2. An [API key](/documentation/qdrant-cloud-api/#authentication-connecting-to-cloud-api) to access the Qdrant cloud API.
|
||||
|
||||
## Example Usage
|
||||
|
||||
The following example creates a new Qdrant cluster in Google Cloud Platform (GCP) and returns the URL of the cluster.
|
||||
|
||||
```terraform
|
||||
terraform {
|
||||
required_version = ">= 1.7.0"
|
||||
required_providers {
|
||||
qdrant-cloud = {
|
||||
source = "qdrant/qdrant-cloud"
|
||||
version = ">=1.1.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
provider "qdrant-cloud" {
|
||||
api_key = "<QDRANT_CLOUD_API_KEY>"
|
||||
account_id = "QDRANT_ACCOUNT_ID>" // Account ID from cloud.qdrant.io/accounts/<QDRANT_ACCOUNT_ID>/ (can be overriden on resource level)
|
||||
}
|
||||
|
||||
resource "qdrant-cloud_accounts_cluster" "example" {
|
||||
name = "tf-example-cluster"
|
||||
cloud_provider = "gcp"
|
||||
cloud_region = "us-east4"
|
||||
configuration {
|
||||
number_of_nodes = 1
|
||||
node_configuration {
|
||||
package_id = "7c939d96-d671-4051-aa16-3b8b7130fa42"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
output "url" {
|
||||
value = qdrant-cloud_accounts_cluster.example.url
|
||||
}
|
||||
```
|
||||
|
||||
The provider includes the following resources and data-sources to work with:
|
||||
|
||||
## Resources
|
||||
|
||||
- `qdrant-cloud_accounts_cluster` - Create clusters on Qdrant cloud - [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/resources/accounts_cluster.md)
|
||||
|
||||
- `qdrant-cloud_accounts_auth_key` - Create API keys for Qdrant cloud clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/resources/accounts_auth_key.md)
|
||||
|
||||
## Data Sources
|
||||
|
||||
- `qdrant-cloud_accounts_auth_keys` - List API keys for Qdrant clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_auth_keys.md)
|
||||
|
||||
- `qdrant-cloud_accounts_cluster` - Get Cluster Information. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_cluster.md)
|
||||
|
||||
- `qdrant-cloud_accounts_clusters` - List Qdrant clusters. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/accounts_clusters.md)
|
||||
|
||||
- `qdrant-cloud_booking_packages` - Get detailed information about the packages/subscriptions available. [Reference](https://github.com/qdrant/terraform-provider-qdrant-cloud/blob/main/docs/data-sources/booking_packages.md)
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Provider Documentation](https://registry.terraform.io/providers/qdrant/qdrant-cloud/latest/docs)
|
||||
- [Terraform Quickstart](https://developer.hashicorp.com/terraform/tutorials)
|
||||
+3
-2
@@ -1,12 +1,13 @@
|
||||
---
|
||||
title: Testcontainers
|
||||
aliases: [ ../frameworks/testcontainers/ ]
|
||||
---
|
||||
|
||||
# Testcontainers
|
||||
|
||||
Qdrant is available as a [Testcontainers module](https://testcontainers.com/modules/qdrant/) in multiple languages. It facilitates the spawning of a Qdrant instance for end-to-end testing.
|
||||
[Testcontainers](https://testcontainers.com/) is a testing library that provides easy and lightweight APIs for bootstrapping integration tests with real services wrapped in Docker containers.
|
||||
|
||||
As noted by [Testcontainers](https://testcontainers.com/), it "is an open source framework for providing throwaway, lightweight instances of databases, message brokers, web browsers, or just about anything that can run in a Docker container."
|
||||
Qdrant is available as a [Testcontainers module](https://testcontainers.com/modules/qdrant/) in multiple languages. It facilitates the spawning of a Qdrant instance for end-to-end testing.
|
||||
|
||||
## Usage
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
---
|
||||
title: Observability
|
||||
weight: 21
|
||||
weight: 22
|
||||
---
|
||||
|
||||
## Observability Integrations
|
||||
|
||||
| Tool | Description |
|
||||
| ----------------------------- | -------------------------------------------------------------------------------------- |
|
||||
| [OpenLIT](./openlit/) | Platform for OpenTelemetry-native Observability & Evals for LLMs and Vector Databases. |
|
||||
| [OpenLLMetry](./openllmetry/) | Set of OpenTelemetry extensions to add Observability for your LLM application. |
|
||||
| [Datadog](./datadog/) | Cloud-based monitoring and analytics platform. |
|
||||
| [OpenLIT](/documentation/observability/openlit/) | Platform for OpenTelemetry-native Observability & Evals for LLMs and Vector Databases. |
|
||||
| [OpenLLMetry](/documentation/observability/openllmetry/) | Set of OpenTelemetry extensions to add Observability for your LLM application. |
|
||||
| [Datadog](/documentation/observability/datadog/) | Cloud-based monitoring and analytics platform. |
|
||||
|
||||
@@ -108,17 +108,17 @@ Let's now evaluate, at a high-level, the way Qdrant is architected.
|
||||
The diagram above represents a high-level overview of some of the main components of Qdrant. Here
|
||||
are the terminologies you should get familiar with.
|
||||
|
||||
- [Collections](../concepts/collections/): A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](../concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements.
|
||||
- [Collections](/documentation/concepts/collections/): A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](/documentation/concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements.
|
||||
- [Distance Metrics](https://en.wikipedia.org/wiki/Metric_space): These are used to measure
|
||||
similarities among vectors and they must be selected at the same time you are creating a
|
||||
collection. The choice of metric depends on the way the vectors were obtained and, in particular,
|
||||
on the neural network that will be used to encode new queries.
|
||||
- [Points](../concepts/points/): The points are the central entity that
|
||||
- [Points](/documentation/concepts/points/): The points are the central entity that
|
||||
Qdrant operates with and they consist of a vector and an optional id and payload.
|
||||
- id: a unique identifier for your vectors.
|
||||
- Vector: a high-dimensional representation of data, for example, an image, a sound, a document, a video, etc.
|
||||
- [Payload](../concepts/payload/): A payload is a JSON object with additional data you can add to a vector.
|
||||
- [Storage](../concepts/storage/): Qdrant can use one of two options for
|
||||
- [Payload](/documentation/concepts/payload/): A payload is a JSON object with additional data you can add to a vector.
|
||||
- [Storage](/documentation/concepts/storage/): Qdrant can use one of two options for
|
||||
storage, **In-memory** storage (Stores all vectors in RAM, has the highest speed since disk
|
||||
access is required only for persistence), or **Memmap** storage, (creates a virtual address
|
||||
space associated with the file on disk).
|
||||
|
||||
@@ -65,7 +65,7 @@ While doing a semantic search at scale, because this is what we sometimes call t
|
||||
Vector search is an exciting alternative to sparse methods. It solves the issues we had with the keyword-based search without needing to maintain lots of heuristics manually. It requires an additional component, a neural encoder, to convert text into vectors.
|
||||
|
||||
[**Tutorial 1 - Qdrant for Complete Beginners**](/documentation/tutorials/search-beginners/)
|
||||
Despite its complicated background, vectors search is extraordinarily simple to set up. With Qdrant, you can have a search engine up-and-running in five minutes. Our [Complete Beginners tutorial](../../tutorials/search-beginners/) will show you how.
|
||||
Despite its complicated background, vectors search is extraordinarily simple to set up. With Qdrant, you can have a search engine up-and-running in five minutes. Our [Complete Beginners tutorial](/documentation/tutorials/search-beginners/) will show you how.
|
||||
|
||||
[**Tutorial 2 - Question and Answer System**](/articles/qa-with-cohere-and-qdrant/)
|
||||
However, you can also choose SaaS tools to generate them and avoid building your model. Setting up a vector search project with Qdrant Cloud and Cohere co.embed API is fairly easy if you follow the [Question and Answer system tutorial](/articles/qa-with-cohere-and-qdrant/).
|
||||
|
||||
@@ -1,19 +1,19 @@
|
||||
---
|
||||
title: Platforms
|
||||
weight: 22
|
||||
weight: 23
|
||||
---
|
||||
|
||||
## Platform Integrations
|
||||
|
||||
| Platform | Description |
|
||||
| ------------------------------------- | ---------------------------------------------------------------------------------------------------- |
|
||||
| [Apify](./apify/) | Platform to build web scrapers and automate web browser tasks. |
|
||||
| [Bubble](./bubble) | Development platform for application development with a no-code interface |
|
||||
| [BuildShip](./buildship) | Low-code visual builder to create APIs, scheduled jobs, and backend workflows. |
|
||||
| [DocsGPT](./docsgpt/) | Tool for ingesting documentation sources and enabling conversations and queries. |
|
||||
| [Make](./make/) | Cloud platform to build low-code workflows by integrating various software applications. |
|
||||
| [N8N](./n8n/) | Platform for node-based, low-code workflow automation. |
|
||||
| [Pipedream](./pipedream/) | Platform for connecting apps and developing event-driven automation. |
|
||||
| [Portable.io](./portable/) | Cloud platform for developing and deploying ELT transformations. |
|
||||
| [PrivateGPT](./privategpt/) | Tool to ask questions about your documents using local LLMs emphasising privacy. |
|
||||
| [Rivet](./rivet/) | A visual programming environment for building AI agents with LLMs. |
|
||||
| --------------------------- | ---------------------------------------------------------------------------------------- |
|
||||
| [Apify](/documentation/platforms/apify/) | Platform to build web scrapers and automate web browser tasks. |
|
||||
| [Bubble](/documentation/platforms/bubble/) | Development platform for application development with a no-code interface |
|
||||
| [BuildShip](/documentation/platforms/buildship/) | Low-code visual builder to create APIs, scheduled jobs, and backend workflows. |
|
||||
| [DocsGPT](/documentation/platforms/docsgpt/) | Tool for ingesting documentation sources and enabling conversations and queries. |
|
||||
| [Make](/documentation/platforms/make/) | Cloud platform to build low-code workflows by integrating various software applications. |
|
||||
| [N8N](/documentation/platforms/n8n/) | Platform for node-based, low-code workflow automation. |
|
||||
| [Pipedream](/documentation/platforms/pipedream/) | Platform for connecting apps and developing event-driven automation. |
|
||||
| [Portable.io](/documentation/platforms/portable/) | Cloud platform for developing and deploying ELT transformations. |
|
||||
| [PrivateGPT](/documentation/platforms/privategpt/) | Tool to ask questions about your documents using local LLMs emphasising privacy. |
|
||||
| [Rivet](/documentation/platforms/rivet/) | A visual programming environment for building AI agents with LLMs. |
|
||||
|
||||
@@ -5,7 +5,7 @@ weight: 14
|
||||
|
||||
# Qdrant Private Cloud
|
||||
|
||||
Qdrant Private Cloud allows you to manage Qdrant database clusters in any Kubernetes cluster on any infrastucture. It uses the same Qdrant Operator that powers Qdrant Managed Cloud and Qdrant Hybrid Cloud, but without any connection to the Qdrant Cloud Management Console.
|
||||
Qdrant Private Cloud allows you to manage Qdrant database clusters in any Kubernetes cluster on any infrastructure. It uses the same Qdrant Operator that powers Qdrant Managed Cloud and Qdrant Hybrid Cloud, but without any connection to the Qdrant Cloud Management Console.
|
||||
|
||||
On top of the open source Qdrant database, it allows
|
||||
|
||||
|
||||
@@ -51,4 +51,4 @@ Use these endpoints to manage your cluster API keys.
|
||||
|
||||
## Terraform Provider
|
||||
|
||||
Qdrant Cloud also provides a Terraform provider to manage your Qdrant Cloud resources. The provider is available on the official [Terraform Registry](https://registry.terraform.io/providers/qdrant/qdrant-cloud).
|
||||
Qdrant Cloud also provides a Terraform provider to manage your Qdrant Cloud resources. [Learn more](/documentation/infrastructure/terraform/).
|
||||
|
||||
@@ -363,7 +363,10 @@ Let's ask a basic question - Which of our stored vectors are most similar to the
|
||||
|
||||
```python
|
||||
search_result = client.query_points(
|
||||
collection_name="test_collection", query=[0.2, 0.1, 0.9, 0.7], limit=3
|
||||
collection_name="test_collection",
|
||||
query=[0.2, 0.1, 0.9, 0.7],
|
||||
with_payload=False,
|
||||
limit=3
|
||||
).points
|
||||
|
||||
print(search_result)
|
||||
@@ -468,7 +471,7 @@ fmt.Println(searchResult)
|
||||
```
|
||||
|
||||
The results are returned in decreasing similarity order. Note that payload and vector data is missing in these results by default.
|
||||
See [payload and vector in the result](../concepts/search/#payload-and-vector-in-the-result) on how to enable it.
|
||||
See [payload and vector in the result](/documentation/concepts/search/#payload-and-vector-in-the-result) on how to enable it.
|
||||
|
||||
## Add a filter
|
||||
|
||||
@@ -591,14 +594,14 @@ fmt.Println(searchResult)
|
||||
]
|
||||
```
|
||||
|
||||
<aside role="status">To make filtered search fast on real datasets, we highly recommend to create <a href="../concepts/indexing/#payload-index">payload indexes</a>!</aside>
|
||||
<aside role="status">To make filtered search fast on real datasets, we highly recommend to create <a href="/documentation/concepts/indexing/#payload-index">payload indexes</a>!</aside>
|
||||
|
||||
You have just conducted vector search. You loaded vectors into a database and queried the database with a vector of your own. Qdrant found the closest results and presented you with a similarity score.
|
||||
|
||||
## Next steps
|
||||
|
||||
Now you know how Qdrant works. Getting started with [Qdrant Cloud](../cloud/quickstart-cloud/) is just as easy. [Create an account](https://qdrant.to/cloud) and use our SaaS completely free. We will take care of infrastructure maintenance and software updates.
|
||||
Now you know how Qdrant works. Getting started with [Qdrant Cloud](/documentation/cloud/quickstart-cloud/) is just as easy. [Create an account](https://qdrant.to/cloud) and use our SaaS completely free. We will take care of infrastructure maintenance and software updates.
|
||||
|
||||
To move onto some more complex examples of vector search, read our [Tutorials](../tutorials/) and create your own app with the help of our [Examples](../examples/).
|
||||
To move onto some more complex examples of vector search, read our [Tutorials](/documentation/tutorials/) and create your own app with the help of our [Examples](/documentation/examples/).
|
||||
|
||||
**Note:** There is another way of running Qdrant locally. If you are a Python developer, we recommend that you try Local Mode in [Qdrant Client](https://github.com/qdrant/qdrant-client), as it only takes a few moments to get setup.
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: Release Notes
|
||||
weight: 29
|
||||
weight: 30
|
||||
type: external-link
|
||||
external_url: https://github.com/qdrant/qdrant/releases
|
||||
sitemapExclude: True
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: Send Data to Qdrant
|
||||
weight: 24
|
||||
weight: 25
|
||||
---
|
||||
|
||||
## How to Send Your Data to a Qdrant Cluster
|
||||
@@ -8,6 +8,6 @@ weight: 24
|
||||
| Example | Description | Stack |
|
||||
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
|
||||
| [Pinecone to Qdrant Data Transfer](https://githubtocolab.com/qdrant/examples/blob/master/data-migration/from-pinecone-to-qdrant.ipynb) | Migrate your vector data from Pinecone to Qdrant. | Qdrant, Vector-io |
|
||||
| [Stream Data to Qdrant with Kafka](../send-data/data-streaming-kafka-qdrant/) | Use Confluent to Stream Data to Qdrant via Managed Kafka. | Qdrant, Kafka |
|
||||
| [Qdrant on Databricks](../send-data/databricks/) | Learn how to use Qdrant on Databricks using the Spark connector | Qdrant, Databricks, Apache Spark |
|
||||
| [Qdrant with Airflow and Astronomer](../send-data/qdrant-airflow-astronomer/) | Build a semantic querying system using Airflow and Astronomer | Qdrant, Airflow, Astronomer |
|
||||
| [Stream Data to Qdrant with Kafka](/documentation/send-data/data-streaming-kafka-qdrant/) | Use Confluent to Stream Data to Qdrant via Managed Kafka. | Qdrant, Kafka |
|
||||
| [Qdrant on Databricks](/documentation/send-data/databricks/) | Learn how to use Qdrant on Databricks using the Spark connector | Qdrant, Databricks, Apache Spark |
|
||||
| [Qdrant with Airflow and Astronomer](/documentation/send-data/qdrant-airflow-astronomer/) | Build a semantic querying system using Airflow and Astronomer | Qdrant, Airflow, Astronomer |
|
||||
|
||||
@@ -14,14 +14,14 @@ These tutorials demonstrate different ways you can build vector search into your
|
||||
|
||||
| Essential How-Tos | Description | Stack |
|
||||
|---------------------------------------------------------------------------------|-------------------------------------------------------------------|---------------------------------------------|
|
||||
| [Semantic Search for Beginners](../tutorials/search-beginners/) | Create a simple search engine locally in minutes. | Qdrant |
|
||||
| [Simple Neural Search](../tutorials/neural-search/) | Build and deploy a neural search that browses startup data. | Qdrant, BERT, FastAPI |
|
||||
| [Neural Search with FastEmbed](../tutorials/neural-search-fastembed/) | Build and deploy a neural search with our FastEmbed library. | Qdrant |
|
||||
| [Multimodal Search](../tutorials/multimodal-search-fastembed/) | Create a simple multimodal search. | Qdrant |
|
||||
| [Bulk Upload Vectors](../tutorials/bulk-upload/) | Upload a large scale dataset. | Qdrant |
|
||||
| [Asynchronous API](../tutorials/async-api/) | Communicate with Qdrant server asynchronously with Python SDK. | Qdrant, Python |
|
||||
| [Create Dataset Snapshots](../tutorials/create-snapshot/) | Turn a dataset into a snapshot by exporting it from a collection. | Qdrant |
|
||||
| [Load HuggingFace Dataset](../tutorials/huggingface-datasets/) | Load a Hugging Face dataset to Qdrant | Qdrant, Python, datasets |
|
||||
| [Measure Retrieval Quality](../tutorials/retrieval-quality/) | Measure and fine-tune the retrieval quality | Qdrant, Python, datasets |
|
||||
| [Search Through Code](../tutorials/code-search/) | Implement semantic search application for code search tasks | Qdrant, Python, sentence-transformers, Jina |
|
||||
| [Setup Collaborative Filtering](../tutorials/collaborative-filtering/) | Implement a collaborative filtering system for recommendation engines | Qdrant|
|
||||
| [Semantic Search for Beginners](/documentation/tutorials/search-beginners/) | Create a simple search engine locally in minutes. | Qdrant |
|
||||
| [Simple Neural Search](/documentation/tutorials/neural-search/) | Build and deploy a neural search that browses startup data. | Qdrant, BERT, FastAPI |
|
||||
| [Neural Search with FastEmbed](/documentation/tutorials/neural-search-fastembed/) | Build and deploy a neural search with our FastEmbed library. | Qdrant |
|
||||
| [Multimodal Search](/documentation/tutorials/multimodal-search-fastembed/) | Create a simple multimodal search. | Qdrant |
|
||||
| [Bulk Upload Vectors](/documentation/tutorials/bulk-upload/) | Upload a large scale dataset. | Qdrant |
|
||||
| [Asynchronous API](/documentation/tutorials/async-api/) | Communicate with Qdrant server asynchronously with Python SDK. | Qdrant, Python |
|
||||
| [Create Dataset Snapshots](/documentation/tutorials/create-snapshot/) | Turn a dataset into a snapshot by exporting it from a collection. | Qdrant |
|
||||
| [Load HuggingFace Dataset](/documentation/tutorials/huggingface-datasets/) | Load a Hugging Face dataset to Qdrant | Qdrant, Python, datasets |
|
||||
| [Measure Retrieval Quality](/documentation/tutorials/retrieval-quality/) | Measure and fine-tune the retrieval quality | Qdrant, Python, datasets |
|
||||
| [Search Through Code](/documentation/tutorials/code-search/) | Implement semantic search application for code search tasks | Qdrant, Python, sentence-transformers, Jina |
|
||||
| [Setup Collaborative Filtering](/documentation/tutorials/collaborative-filtering/) | Implement a collaborative filtering system for recommendation engines | Qdrant|
|
||||
|
||||
@@ -101,23 +101,23 @@ client.updateCollection("{collection_name}", {
|
||||
## Upload directly to disk
|
||||
|
||||
When the vectors you upload do not all fit in RAM, you likely want to use
|
||||
[memmap](../../concepts/storage/#configuring-memmap-storage)
|
||||
[memmap](/documentation/concepts/storage/#configuring-memmap-storage)
|
||||
support.
|
||||
|
||||
During collection
|
||||
[creation](../../concepts/collections/#create-collection),
|
||||
[creation](/documentation/concepts/collections/#create-collection),
|
||||
memmaps may be enabled on a per-vector basis using the `on_disk` parameter. This
|
||||
will store vector data directly on disk at all times. It is suitable for
|
||||
ingesting a large amount of data, essential for the billion scale benchmark.
|
||||
|
||||
Using `memmap_threshold_kb` is not recommended in this case. It would require
|
||||
the [optimizer](../../concepts/optimizer/) to constantly
|
||||
the [optimizer](/documentation/concepts/optimizer/) to constantly
|
||||
transform in-memory segments into memmap segments on disk. This process is
|
||||
slower, and the optimizer can be a bottleneck when ingesting a large amount of
|
||||
data.
|
||||
|
||||
Read more about this in
|
||||
[Configuring Memmap Storage](../../concepts/storage/#configuring-memmap-storage).
|
||||
[Configuring Memmap Storage](/documentation/concepts/storage/#configuring-memmap-storage).
|
||||
|
||||
## Parallel upload into multiple shards
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@ algorithm used in Qdrant, to obtain the best results.
|
||||
The quality of the embeddings is a topic for a separate tutorial. In a nutshell, it is usually measured and compared by benchmarks, such as
|
||||
[Massive Text Embedding Benchmark (MTEB)](https://huggingface.co/spaces/mteb/leaderboard). The evaluation process itself is pretty
|
||||
straightforward and is based on a ground truth dataset built by humans. We have a set of queries and a set of the documents we would expect
|
||||
to receive for each of them. In the evaluation process, we take a query, find the most similar documents in the vector space and compare
|
||||
to receive for each of them. In the [evaluation process](https://qdrant.tech/rag/rag-evaluation-guide/), we take a query, find the most similar documents in the vector space and compare
|
||||
them with the ground truth. In that setup, **finding the most similar documents is implemented as full kNN search, without any approximation**.
|
||||
As a result, we can measure the quality of the embeddings themselves, without the influence of the ANN algorithm.
|
||||
|
||||
@@ -50,7 +50,7 @@ algorithm approximates the exact search**.
|
||||
|
||||
## Measure the quality of the search results
|
||||
|
||||
Let's build a quality evaluation of the ANN algorithm in Qdrant. We will, first, call the search endpoint in a standard way to obtain
|
||||
Let's build a quality [evaluation](https://qdrant.tech/rag/rag-evaluation-guide/) of the ANN algorithm in Qdrant. We will, first, call the search endpoint in a standard way to obtain
|
||||
the approximate search results. Then, we will call the exact search endpoint to obtain the exact matches, and finally compare both results
|
||||
in terms of precision.
|
||||
|
||||
@@ -218,7 +218,7 @@ to do it.
|
||||
|
||||
## Wrapping up
|
||||
|
||||
Assessing the quality of retrieval is a critical aspect of evaluating semantic search performance. It is imperative to measure retrieval quality when aiming for optimal quality of.
|
||||
Assessing the quality of retrieval is a critical aspect of [evaluating](https://qdrant.tech/rag/rag-evaluation-guide/) semantic search performance. It is imperative to measure retrieval quality when aiming for optimal quality of.
|
||||
your search results. Qdrant provides a built-in exact search mode, which can be used to measure the quality of the ANN algorithm itself,
|
||||
even in an automated way, as part of your CI/CD pipeline.
|
||||
|
||||
|
||||
@@ -240,7 +240,7 @@ The query has been narrowed down to one result from 2008.
|
||||
|
||||
## Next Steps
|
||||
|
||||
Congratulations, you have just created your very first search engine! Trust us, the rest of Qdrant is not that complicated, either. For your next tutorial you should try building an actual [Neural Search Service with a complete API and a dataset](../../tutorials/neural-search/).
|
||||
Congratulations, you have just created your very first search engine! Trust us, the rest of Qdrant is not that complicated, either. For your next tutorial you should try building an actual [Neural Search Service with a complete API and a dataset](/documentation/tutorials/neural-search/).
|
||||
|
||||
## Return to the bash shell
|
||||
|
||||
|
||||
@@ -96,6 +96,9 @@ menuItems:
|
||||
- id: 2
|
||||
name: Articles
|
||||
url: /articles/
|
||||
- id: 3
|
||||
name: Startup Program
|
||||
url: /qdrant-for-startups/
|
||||
- title: Company
|
||||
items:
|
||||
- id: 0
|
||||
@@ -127,5 +130,8 @@ bages:
|
||||
- src: /img/soc2-badge.png
|
||||
alt: "SOC2"
|
||||
url: http://qdrant.to/trust-center
|
||||
- src: /img/gdpr-badge.png
|
||||
alt: "heyData GDPR"
|
||||
url: https://heydata.eu/
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -99,6 +99,10 @@ menuItems:
|
||||
name: Articles
|
||||
icon: articles.svg
|
||||
url: /articles/
|
||||
- id: subMenu-3-3
|
||||
name: Startup Program
|
||||
icon: qdrant-for-startups.svg
|
||||
url: /qdrant-for-startups/
|
||||
- id: menu-4
|
||||
name: Company
|
||||
subMenuItems:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
stats:
|
||||
githubStars: 19.8k
|
||||
discordMembers: 6.5k
|
||||
githubStars: 20.0k
|
||||
discordMembers: 6.6k
|
||||
twitterFollowers: 7.5k
|
||||
---
|
||||
@@ -10,11 +10,11 @@ icon: <svg width="16" height="16" viewBox="0 0 16 16" fill="none"
|
||||
7.86204 15.74L14.5287 7.07333C14.6834 6.87199 14.71 6.59999 14.598 6.37199Z"
|
||||
fill="#8547FF"/></g><defs><clipPath id="clip0_770_2716"><rect width="16"
|
||||
height="16" fill="white"/></clipPath></defs></svg>
|
||||
text: "Webinar: Building Agents with LlamaIndex & Qdrant"
|
||||
text: "Guide: Best Practices in RAG Evaluation"
|
||||
link:
|
||||
text: Register now
|
||||
url: https://try.qdrant.tech/build-advanced-agents-with-llamaindex-and-qdrant?utm_source=website&utm_medium=homepage-banner&utm_campaign=sept-webinar-agents-llamaindex
|
||||
start: 2024-09-07T15:28:00.000Z
|
||||
text: Read now
|
||||
url: https://qdrant.tech/rag/rag-evaluation-guide/?utm_source=website&utm_medium=homepage-banner&utm_campaign=rag-eval-guide
|
||||
start: 2024-10-01T15:28:00.000Z
|
||||
sitemapExclude: true
|
||||
end: 2024-09-26T08:00:00.000Z
|
||||
end: 2024-10-31T08:00:00.000Z
|
||||
---
|
||||
|
||||
@@ -1,13 +1,11 @@
|
||||
---
|
||||
title: Qdrant For Startups
|
||||
description: Qdrant For Startups
|
||||
cascade:
|
||||
- _target:
|
||||
environment: production
|
||||
title: Qdrant for Startups
|
||||
description: Supporting early-stage startups with discounts on Qdrant Cloud, technical guidance, and access to key AI tools from LlamaIndex, Hugging Face, and Airbyte.
|
||||
build:
|
||||
list: never
|
||||
render: never
|
||||
render: always
|
||||
cascade:
|
||||
- build:
|
||||
list: local
|
||||
publishResources: false
|
||||
sitemapExclude: true
|
||||
# todo: remove sitemapExclude and change building options after the page is ready to be published
|
||||
render: never
|
||||
---
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
title: Why join Qdrant for Startups?
|
||||
mainCard:
|
||||
title: Discount for Qdrant Cloud
|
||||
description: Receive a discount on <a href="https://cloud.qdrant.io/" target="_blank">Qdrant Cloud</a> for the first year.
|
||||
description: Enjoy a discount on <a href="https://cloud.qdrant.io/" target="_blank">Qdrant Cloud</a> for the first year.
|
||||
image:
|
||||
src: /img/qdrant-for-startups-benefits/card1.png
|
||||
alt: Qdrant Discount for Startups
|
||||
@@ -14,14 +14,13 @@ cards:
|
||||
src: /img/qdrant-for-startups-benefits/card2.svg
|
||||
alt: Expert Technical Advice
|
||||
- id: 1
|
||||
title: Co-Marketing Opportunities
|
||||
description: We’d love to share your work with our community. Exclusive access to our Vector Space Talks, joint blog posts, and more.
|
||||
title: Partner Perks
|
||||
description: Receive exclusive perks from Hugging Face, LlamaIndex, and Airbyte, ensuring you have access to key tools and resources for AI-driven applications.
|
||||
image:
|
||||
src: /img/qdrant-for-startups-benefits/card3.svg
|
||||
alt: Co-Marketing Opportunities
|
||||
description: Qdrant is the leading open source vector database and similarity search engine designed to handle high-dimensional vectors for performance and massive-scale AI applications.
|
||||
link:
|
||||
url: /documentation/overview/
|
||||
text: Learn More
|
||||
button:
|
||||
text: Apply Now
|
||||
url: "#form"
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -4,37 +4,38 @@ questions:
|
||||
- id: 0
|
||||
question: What are the eligibility requirements?
|
||||
answer: |
|
||||
<p>You must meet all of the following:</p>
|
||||
<ul>
|
||||
<li>Pre-seed, Seed or Series A startups (under five years old)</li>
|
||||
<li>New user of Qdrant Cloud</li>
|
||||
<li>Not a previous participant in the Qdrant for Startups program</li>
|
||||
<li>Be building an AI-driven product or service (agencies or devshops are not eligible)</li>
|
||||
<li>Have a live, functional website</li>
|
||||
<li>Billing must be done directly with Qdrant (not through a marketplace)</li>
|
||||
<li>New user of Qdrant Cloud.</li>
|
||||
<li>Pre-seed, Seed, or Series A startups (under five years old) and less than $5M in funding.</li>
|
||||
<li>Have not previously participated in the Qdrant for Startups program</li>
|
||||
<li>Building an AI-driven product or services (agencies or devshops are not eligible)</li>
|
||||
<li>Provide a link to a live, functional website</li>
|
||||
<li>Billing will be done directly with Qdrant (not through a marketplace)</li>
|
||||
</ul>
|
||||
- id: 1
|
||||
question: When will I get notified about my application?
|
||||
answer: Upon submitting your application, we will review it and notify you of your status within 7 business days.
|
||||
question: How can I apply to the Qdrant Startup Program?
|
||||
answer: Apply through our online form by providing details about your startup and plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities.
|
||||
- id: 2
|
||||
question: What is the price?
|
||||
answer: It is free to apply to the program. As part of the program, you will receive a discount on Qdrant Cloud, valid for 12 months. For detailed cloud pricing, please visit qdrant.tech/pricing.
|
||||
- id: 3
|
||||
question: How can my startup join the program?
|
||||
answer: Your startup can join the program by simply submitting the application on this page. Once submitted, we will review your application and notify you of your status within 7 business days.
|
||||
- id: 4
|
||||
question: What criteria are used to select startups for the program?
|
||||
answer: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected.
|
||||
- id: 3
|
||||
question: How long is the discount valid, and what are the conditions?
|
||||
answer: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount. Billing can not be through a marketplace. For details on pricing, please visit qdrant.tech/pricing.
|
||||
- id: 4
|
||||
question: How can I maximize the co-marketing opportunities offered by the program?
|
||||
answer: Engage actively with our marketing team for features on social media, possible appearances in Discord talks or webinars, and case studies to maximize your startup's visibility and showcase your innovative use of Qdrant.
|
||||
- id: 5
|
||||
question: How long is the discount valid, and are there any conditions?
|
||||
answer: The discount is valid for 12 months from the date of acceptance and applies exclusively to our Cloud services billed through Stripe. Participants need a Stripe account to utilize the discount.
|
||||
- id: 6
|
||||
question: Can existing Qdrant customers apply for the Startup Program?
|
||||
answer: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered by the program.
|
||||
- id: 7
|
||||
answer: Yes, existing Qdrant customers are eligible to apply for the Startup Program if their cloud account was created within the last 30 days from the date of application. This opportunity is designed to ensure startups at the early stages of using our platform can still benefit from the additional support and resources offered.
|
||||
- id: 6
|
||||
question: Can I reapply if my application is initially rejected?
|
||||
answer: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply.
|
||||
- id: 8
|
||||
- id: 7
|
||||
question: Who can I contact for more information about the program?
|
||||
answer: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com.
|
||||
answer: After reading these FAQs in full, if you need more details or assistance, please contact <a href="mailto:startups@qdrant.com">startups@qdrant.com</a>.
|
||||
button:
|
||||
text: Apply Now
|
||||
url: "#form"
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -1,12 +1,11 @@
|
||||
---
|
||||
title: Qdrant for Startups
|
||||
description: Powering The Next Wave of AI Innovators, Qdrant for Startups is committed to being the catalyst for the next generation of AI pioneers. Our program is specifically designed to provide AI-focused startups with the right resources to scale. If AI is at the heart of your startup, you're in the right place.
|
||||
description: Powering The Next Wave of AI Innovators, Qdrant for Startups is committed to being the catalyst for the next generation of AI pioneers.<br><br>Our program is specifically designed to provide AI-focused startups with the right resources to scale. If AI is at the heart of your startup, you're in the right place.
|
||||
button:
|
||||
text: Apply Now
|
||||
url: "#form"
|
||||
image:
|
||||
src: /img/qdrant-for-startups-hero.svg
|
||||
srcMobile: /img/mobile/qdrant-for-startups-hero.svg
|
||||
src: /img/startups-program.svg
|
||||
alt: Qdrant for Startups
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
+2
-2
@@ -9,8 +9,8 @@ features:
|
||||
title: Question and Answer System with LlamaIndex
|
||||
description: Combine Qdrant and LlamaIndex to create a self-updating Q&A system.
|
||||
link:
|
||||
text: Video Tutorial
|
||||
url: https://www.youtube.com/watch?v=id5ql-Abq4Y&t=56s
|
||||
text: Jupyter Notebook
|
||||
url: https://github.com/qdrant/examples/blob/949669f001a03131afebf2ecd1e0ce63cab01c81/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb
|
||||
- id: 1
|
||||
image:
|
||||
src: /img/retrieval-augmented-generation-use-cases/case2.svg
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
<h2>Confirm your signup</h2>
|
||||
|
||||
<p>Follow this link to confirm your user:</p>
|
||||
<p><a href="{{ .SiteURL }}/admin/#confirmation_token={{ .Token }}">Confirm your mail</a></p>
|
||||
<p><a href="http://{{ .SiteURL }}/admin/#confirmation_token={{ .Token }}">Confirm your mail</a></p>
|
||||
|
||||
@@ -4,4 +4,4 @@
|
||||
Follow this link to confirm the update of your email from
|
||||
{{ .Email }} to {{ .NewEmail }}:
|
||||
</p>
|
||||
<p><a href="{{ .SiteURL }}/admin/#email_change_token={{ .Token }}">Change Email</a></p>
|
||||
<p><a href="http://{{ .SiteURL }}/admin/#email_change_token={{ .Token }}">Change Email</a></p>
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user