mirror of
https://github.com/qdrant/landing_page.git
synced 2026-09-28 23:48:31 +02:00
Merge branch 'master' into q-and-a-update
This commit is contained in:
@@ -14,9 +14,13 @@ jobs:
|
||||
- name: Setup Hugo
|
||||
uses: peaceiris/actions-hugo@v2
|
||||
with:
|
||||
hugo-version: "latest"
|
||||
hugo-version: "latest"
|
||||
- name: Install npm
|
||||
uses: actions/setup-node@v2
|
||||
with:
|
||||
node-version: '20'
|
||||
- name: Run hugo
|
||||
run: cd qdrant-landing && hugo -b 'https://qdrant.tech/'
|
||||
run: bash -x ./install-and-build.sh
|
||||
- name: Link Checker
|
||||
id: lychee
|
||||
uses: lycheeverse/lychee-action@v1.9.1
|
||||
|
||||
@@ -15,9 +15,16 @@ jobs:
|
||||
- name: Setup Hugo
|
||||
uses: peaceiris/actions-hugo@v2
|
||||
with:
|
||||
hugo-version: "latest"
|
||||
hugo-version: "0.123.0"
|
||||
- name: Install npm
|
||||
uses: actions/setup-node@v2
|
||||
with:
|
||||
node-version: '20'
|
||||
- name: Run hugo
|
||||
run: |
|
||||
bash -x ./install-and-build.sh
|
||||
CURRENT_DIR=$(pwd)
|
||||
export PATH="${CURRENT_DIR}/dart-sass:${PATH}"
|
||||
cd qdrant-landing && hugo --gc -b 'http://localhost:1313' && hugo serve &
|
||||
sleep 5 # wait for server to start
|
||||
- name: Link Checker
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
name: Update Github Stars
|
||||
|
||||
on:
|
||||
repository_dispatch:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "00 17 * * 0"
|
||||
|
||||
jobs:
|
||||
update:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Run github-stars update script
|
||||
run: |
|
||||
bash -x automation/update-stats.sh
|
||||
- uses: stefanzweifel/git-auto-commit-action@v5
|
||||
with:
|
||||
commit_message: "Update GitHub Stars"
|
||||
commit_user_name: "GitHub Actions"
|
||||
commit_user_email: "team@qdrant.com"
|
||||
@@ -10,7 +10,7 @@
|
||||
|
||||
- [Node.js](https://nodejs.org/en/download/)
|
||||
- [npm](https://www.npmjs.com/get-npm)
|
||||
- [sass](https://sass-lang.com/install)
|
||||
- [sass](https://sass-lang.com/install) - be aware that you need a Dart Sass version, don't use npm package `sass` as it's a different implementation of Sass
|
||||
|
||||
## Run
|
||||
|
||||
@@ -34,21 +34,7 @@ hugo serve -D
|
||||
|
||||
## Build css from scss
|
||||
|
||||
If you are **going to change scss files**, you need to run the following commands in a separate terminal window.
|
||||
|
||||
Install sass if you don't have it:
|
||||
|
||||
```bash
|
||||
npm install -g sass
|
||||
```
|
||||
|
||||
Install dependencies and run sass watcher:
|
||||
|
||||
``` bash
|
||||
cd qdrant-landing
|
||||
npm install
|
||||
sass --watch --style=compressed ./themes/qdrant/static/css/main.scss ./themes/qdrant/static/css/main.css
|
||||
```
|
||||
For previous theme, it was required to build css files. We don't need to explicitly build css from scss anymore. It's done automatically by Hugo uses Dart Sass, which should be installed on your machine to see results).
|
||||
|
||||
# Content Management
|
||||
|
||||
@@ -263,7 +249,7 @@ In the blog post file, you'll see:
|
||||
|
||||
- Add tags. While they're not shown on the blog post page, they are used to display related posts.
|
||||
- If post has `featured: true` property in the front matter this post will appear in the "Features and News" blog section. Only the last 4 featured posts will be displayed in this section. Featured posts will not appear in the regular post list.
|
||||
- If there are more than 4 `featured: true` posts (where `draft: false`), the oldest post disappears from https://qdrant.tech/blog.
|
||||
- If there are more than 4 `featured: true` posts (where `draft: false`), the oldest post disappears from /blog.
|
||||
|
||||
## Marketing Landing Pages
|
||||
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e
|
||||
|
||||
STAR_COUNT=$(curl -s https://api.github.com/repos/qdrant/qdrant | jq '.stargazers_count')
|
||||
DISCORD_COUNT=$(curl -s https://discord.com/api/v9/invites/qdrant?with_counts=true&with_expiration=true)
|
||||
DISCORD_COUNT=$(echo $DISCORD_COUNT | jq '.approximate_member_count')
|
||||
|
||||
if [ -z "$STAR_COUNT" ]; then
|
||||
echo "Failed to get the star count"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# humanize the stats count like 18322 -> 18.3k
|
||||
# and if star count is bigger than 1000, then add k at the end
|
||||
STAR_COUNT=$(echo $STAR_COUNT | awk '{ if($1 > 1000) { printf "%.1fk\n", $1/1000 } else { print $1 } }' | tr ',' '.')
|
||||
DISCORD_COUNT=$(echo $DISCORD_COUNT | awk '{ if($1 > 1000) { printf "%.1fk\n", $1/1000 } else { print $1 } }' | tr ',' '.')
|
||||
|
||||
# Update the star count in the markdown file
|
||||
echo "Current star count: $STAR_COUNT"
|
||||
echo "Updating the star count in the stats.md file"
|
||||
sed -i "s/.*githubStars.*$/ githubStars: $STAR_COUNT/g" ./qdrant-landing/content/headless/stats.md
|
||||
echo "Current discord count: $DISCORD_COUNT"
|
||||
echo "Updating the discord count in the stats.md file"
|
||||
sed -i "s/.*discordMembers.*$/ discordMembers: $DISCORD_COUNT/g" ./qdrant-landing/content/headless/stats.md
|
||||
@@ -0,0 +1,13 @@
|
||||
#!/bin/bash
|
||||
|
||||
DART_SASS_VERSION=${DART_SASS_VERSION:-1.70.0}
|
||||
DEPLOY_PRIME_URL=${DEPLOY_PRIME_URL:-"https://qdrant.com"}
|
||||
|
||||
CURRENT_DIR=$(pwd)
|
||||
|
||||
curl -LJO https://github.com/sass/dart-sass/releases/download/${DART_SASS_VERSION}/dart-sass-${DART_SASS_VERSION}-linux-x64.tar.gz && \
|
||||
tar -xf dart-sass-${DART_SASS_VERSION}-linux-x64.tar.gz && \
|
||||
rm dart-sass-${DART_SASS_VERSION}-linux-x64.tar.gz && \
|
||||
export PATH="${CURRENT_DIR}/dart-sass:${PATH}" && \
|
||||
cd qdrant-landing && npm install && hugo --gc --minify --config config.toml,config-theme.toml --buildFuture -b ${DEPLOY_PRIME_URL}
|
||||
|
||||
+4
-4
@@ -1,20 +1,20 @@
|
||||
[build]
|
||||
publish = "qdrant-landing/public"
|
||||
command = "cd qdrant-landing ; hugo --gc --minify"
|
||||
command = "bash -x ./install-and-build.sh"
|
||||
|
||||
[build.environment]
|
||||
HUGO_VERSION = "0.123.0"
|
||||
DART_SASS_VERSION = "1.70.0"
|
||||
NODE_VERSION = "20.10.0"
|
||||
|
||||
[context.production.environment]
|
||||
HUGO_ENV = "production"
|
||||
HUGO_ENABLEGITINFO = "true"
|
||||
DEPLOY_PRIME_URL = "https://qdrant.tech"
|
||||
|
||||
[context.deploy-preview.environment]
|
||||
HUGO_ENV = "staging"
|
||||
|
||||
[context.deploy-preview]
|
||||
command = "cd qdrant-landing ; echo $HUGO_ENV ; hugo --gc --minify --buildFuture -b $DEPLOY_PRIME_URL"
|
||||
|
||||
[[headers]]
|
||||
for = "/*"
|
||||
[headers.values]
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
---
|
||||
title: "{{ replace .Name "-" " " | title }}"
|
||||
draft: false
|
||||
slug: {{ .Name }} # Change this slug to your page slug if needed
|
||||
short_description: This is a blog post # Change this
|
||||
description: This is a blog post # Change this
|
||||
preview_image: /blog/Article-Image.png # Change this
|
||||
|
||||
# social_preview_image: /blog/Article-Image.png # Optional image used for link previews
|
||||
# title_preview_image: /blog/Article-Image.png # Optional image used for blog post title
|
||||
# small_preview_image: /blog/Article-Image.png # Optional image used for small preview in the list of blog posts
|
||||
|
||||
date: {{ .Date }}
|
||||
author: John Doe # Change this
|
||||
featured: false # if true, this post will be featured on the blog page
|
||||
tags: # Change this, related by tags posts will be shown on the blog page
|
||||
- news
|
||||
- blog
|
||||
weight: 0 # Change this weight to change order of posts
|
||||
# For more guidance, see https://github.com/qdrant/landing_page?tab=readme-ov-file#blog
|
||||
---
|
||||
|
||||
Here is your blog post content. You can use markdown syntax here.
|
||||
|
||||
# Header 1
|
||||
## Header 2
|
||||
### Header 3
|
||||
#### Header 4
|
||||
##### Header 5
|
||||
###### Header 6
|
||||
|
||||
<aside role="alert">
|
||||
You can add a note to your page using this aside block.
|
||||
</aside>
|
||||
|
||||
<aside role="status">
|
||||
This is a warning message.
|
||||
</aside>
|
||||
|
||||
> This is a blockquote following a header.
|
||||
|
||||
Table:
|
||||
|
||||
| Header 1 | Header 2 | Header 3 | Header 4 |
|
||||
| -------- | -------- | -------- | -------- |
|
||||
| Cell 1 | Cell 2 | Cell 3 | Cell 4 |
|
||||
| Cell 3 | Cell 4 | Cell 5 | Cell 6 |
|
||||
|
||||
- List item 1
|
||||
- Nested list item 1
|
||||
- Nested list item 2
|
||||
- List item 2
|
||||
- List item 3
|
||||
|
||||
1. Numbered list item 1
|
||||
1. Nested numbered list item 1
|
||||
2. Nested numbered list item 2
|
||||
2. Numbered list item 2
|
||||
3. Numbered list item 3
|
||||
@@ -0,0 +1,7 @@
|
||||
---
|
||||
#Delimiter files are used to separate the list of documentation pages into sections.
|
||||
title: "{{ replace .Name "-" " " | title }}"
|
||||
type: delimiter
|
||||
weight: 0 # Change this weight to change order of sections
|
||||
sitemapExclude: True
|
||||
---
|
||||
@@ -0,0 +1,7 @@
|
||||
---
|
||||
# External link template
|
||||
title: "{{ replace .Name "-" " " | title }}"
|
||||
type: external-link
|
||||
external_url: https://github.com/qdrant/qdrant # Change this link to your external link
|
||||
sitemapExclude: True
|
||||
---
|
||||
@@ -0,0 +1 @@
|
||||
theme = "qdrant-2024"
|
||||
@@ -1,7 +1,7 @@
|
||||
baseURL = "https://qdrant.tech"
|
||||
languageCode = "en-us"
|
||||
title = "Qdrant - Vector Database"
|
||||
theme = "qdrant"
|
||||
theme = "qdrant-2024"
|
||||
|
||||
googleAnalytics = "G-NZYW2651NE"
|
||||
enableRobotsTXT = true
|
||||
@@ -80,6 +80,7 @@ disableKinds = ["taxonomy", "term"]
|
||||
cloudDocVersion = "v0.1.x"
|
||||
|
||||
googleTagManager = "GTM-KRLCXD5"
|
||||
segmentWriteKey = ""
|
||||
|
||||
[params.author]
|
||||
name = "Andrey Vasnetsov"
|
||||
@@ -314,3 +315,14 @@ disableKinds = ["taxonomy", "term"]
|
||||
[[related.indices]]
|
||||
name = "tags"
|
||||
weight = 100
|
||||
|
||||
# Not rendering the content of the cd following files in production
|
||||
[[cascade]]
|
||||
[cascade._build]
|
||||
list = 'never'
|
||||
render = 'never'
|
||||
[cascade._target]
|
||||
environment = '{production}'
|
||||
path = '{**.skip,**.skip/*}'
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
---
|
||||
title: About Us
|
||||
---
|
||||
@@ -0,0 +1,11 @@
|
||||
---
|
||||
title: advanced-search
|
||||
description: advanced-search
|
||||
build:
|
||||
render: always
|
||||
cascade:
|
||||
- build:
|
||||
list: local
|
||||
publishResources: false
|
||||
render: never
|
||||
---
|
||||
@@ -0,0 +1,43 @@
|
||||
---
|
||||
title: Search with Qdrant
|
||||
description: Qdrant enhances search, offering semantic, similarity, multimodal, and hybrid search capabilities for accurate, user-centric results, serving applications in different industries like e-commerce to healthcare.
|
||||
features:
|
||||
- id: 0
|
||||
icon:
|
||||
src: /icons/outline/similarity-blue.svg
|
||||
alt: Similarity
|
||||
title: Semantic Search
|
||||
description: Qdrant optimizes similarity search, identifying the closest database items to any query vector for applications like recommendation systems, RAG and image retrieval, enhancing accuracy and user experience.
|
||||
link:
|
||||
text: Learn More
|
||||
url: /documentation/concepts/search/
|
||||
- id: 1
|
||||
icon:
|
||||
src: /icons/outline/search-text-blue.svg
|
||||
alt: Search text
|
||||
title: Hybrid Search for Text
|
||||
description: By combining dense vector embeddings with sparse vectors e.g. BM25, Qdrant powers semantic search to deliver context-aware results, transcending traditional keyword search by understanding the deeper meaning of data.
|
||||
link:
|
||||
text: Learn More
|
||||
url: /documentation/tutorials/hybrid-search-fastembed/
|
||||
- id: 2
|
||||
icon:
|
||||
src: /icons/outline/selection-blue.svg
|
||||
alt: Selection
|
||||
title: Multimodal Search
|
||||
description: Qdrant's capability extends to multi-modal search, indexing and retrieving various data forms (text, images, audio) once vectorized, facilitating a comprehensive search experience.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: /documentation/tutorials/aleph-alpha-search/
|
||||
- id: 3
|
||||
icon:
|
||||
src: /icons/outline/filter-blue.svg
|
||||
alt: Filter
|
||||
title: Single Stage filtering that Works
|
||||
description: Qdrant enhances search speeds and control and context understanding through filtering on any nested entry in our payload. Unique architecture allows Qdrant to avoid expensive pre-filtering and post-filtering stages, making search faster and accurate.
|
||||
link:
|
||||
text: Learn More
|
||||
url: /articles/filtrable-hnsw/
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
---
|
||||
title: Advanced Search
|
||||
description: Dive into next-gen search capabilities with Qdrant, offering a smarter way to deliver precise and tailored content to users, enhancing interaction accuracy and depth.
|
||||
startFree:
|
||||
text: Get Started
|
||||
url: https://cloud.qdrant.io/
|
||||
learnMore:
|
||||
text: Contact Us
|
||||
url: /contact-us/
|
||||
image:
|
||||
src: /img/vectors/vector-0.svg
|
||||
alt: Advanced search
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
---
|
||||
title: Learn how to get started with Qdrant for your search use case
|
||||
features:
|
||||
- id: 0
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/startup-semantic-search.svg
|
||||
alt: Startup Semantic Search
|
||||
title: Startup Semantic Search Demo
|
||||
description: The demo showcases semantic search for startup descriptions through SentenceTransformer and Qdrant, comparing neural search's accuracy with traditional searches for better content discovery.
|
||||
link:
|
||||
text: View Demo
|
||||
url: https://demo.qdrant.tech/
|
||||
- id: 1
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/multimodal-semantic-search.svg
|
||||
alt: Multimodal Semantic Search
|
||||
title: Multimodal Semantic Search with Aleph Alpha
|
||||
description: This tutorial shows you how to run a proper multimodal semantic search system with a few lines of code, without the need to annotate the data or train your networks.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: /documentation/examples/aleph-alpha-search/
|
||||
- id: 2
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/simple-neural-search.svg
|
||||
alt: Simple Neural Search
|
||||
title: Create a Simple Neural Search Service
|
||||
description: This tutorial shows you how to build and deploy your own neural search service.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: /documentation/tutorials/neural-search/
|
||||
- id: 3
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/image-classification.svg
|
||||
alt: Image Classification
|
||||
title: Image Classification with Qdrant Vector Semantic Search
|
||||
description: In this tutorial, you will learn how a semantic search engine for images can help diagnose different types of skin conditions.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: https://www.youtube.com/watch?v=sNFmN16AM1o
|
||||
- id: 4
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/semantic-search-101.svg
|
||||
alt: Semantic Search 101
|
||||
title: Semantic Search 101
|
||||
description: Build a semantic search engine for science fiction books in 5 mins.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: /documentation/tutorials/search-beginners/
|
||||
- id: 5
|
||||
image:
|
||||
src: /img/advanced-search-use-cases/hybrid-search-service-fastembed.svg
|
||||
alt: Create a Hybrid Search Service with Fastembed
|
||||
title: Create a Hybrid Search Service with Fastembed
|
||||
description: This tutorial guides you through building and deploying your own hybrid search service using Fastembed.
|
||||
link:
|
||||
text: View Tutorial
|
||||
url: /documentation/tutorials/hybrid-search-fastembed/
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
---
|
||||
title: Articles
|
||||
title: Qdrant Articles
|
||||
page_title: Articles about Vector Search
|
||||
description: Articles about vector search and similarity larning related topics. Latest updates on Qdrant vector search engine.
|
||||
section_title: Check out our latest publications
|
||||
subtitle: Check out our latest publications
|
||||
img: /articles_data/title-img.png
|
||||
---
|
||||
---
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
---
|
||||
title: "Enhance OpenAI Embeddings with Qdrant's Binary Quantization"
|
||||
title: "Optimizing OpenAI Embeddings: Enhance Efficiency with Qdrant's Binary Quantization"
|
||||
draft: false
|
||||
slug: binary-quantization-openai
|
||||
short_description: Use Qdrant's Binary Quantization to enhance OpenAI embeddings
|
||||
description: Use Qdrant's Binary Quantization to enhance the performance and efficiency of OpenAI embeddings
|
||||
description: Explore how Qdrant's Binary Quantization can significantly improve the efficiency and performance of OpenAI's Ada-003 embeddings. Learn best practices for real-time search applications.
|
||||
preview_dir: /articles_data/binary-quantization-openai/preview
|
||||
preview_image: /articles-data/binary-quantization-openai/Article-Image.png
|
||||
small_preview_image: /articles_data/binary-quantization-openai/icon.svg
|
||||
@@ -38,19 +38,19 @@ If you're new to Binary Quantization, consider reading our article which walks y
|
||||
|
||||
You can also try out these techniques as described in [Binary Quantization OpenAI](https://github.com/qdrant/examples/blob/openai-3/binary-quantization-openai/README.md), which includes Jupyter notebooks.
|
||||
|
||||
## New OpenAI Embeddings: Performance and Changes
|
||||
## New OpenAI embeddings: performance and changes
|
||||
|
||||
As the technology of embedding models has advanced, demand has grown. Users are looking more for powerful and efficient text-embedding models. OpenAI's Ada-003 embeddings offer state-of-the-art performance on a wide range of NLP tasks, including those noted in [MTEB](https://huggingface.co/spaces/mteb/leaderboard) and [MIRACL](https://openai.com/blog/new-embedding-models-and-api-updates).
|
||||
|
||||
These models include multilingual support in over 100 languages. The transition from text-embedding-ada-002 to text-embedding-3-large has led to a significant jump in performance scores (from 31.4% to 54.9% on MIRACL).
|
||||
|
||||
#### Matryoshka Representation Learning
|
||||
#### Matryoshka representation learning
|
||||
|
||||
The new OpenAI models have been trained with a novel approach called "[Matryoshka Representation Learning](https://aniketrege.github.io/blog/2024/mrl/)". Developers can set up embeddings of different sizes (number of dimensions). In this post, we use small and large variants. Developers can select embeddings which balances accuracy and size.
|
||||
|
||||
Here, we show how the accuracy of binary quantization is quite good across different dimensions -- for both the models.
|
||||
|
||||
## Enhanced Performance and Efficiency with Binary Quantization
|
||||
## Enhanced performance and efficiency with binary quantization
|
||||
|
||||
By reducing storage needs, you can scale applications with lower costs. This addresses a critical challenge posed by the original embedding sizes. Binary Quantization also speeds the search process. It simplifies the complex distance calculations between vectors into more manageable bitwise operations, which supports potentially real-time searches across vast datasets.
|
||||
|
||||
@@ -64,7 +64,7 @@ The efficiency gains from Binary Quantization are as follows:
|
||||
- Enhanced speed of data retrieval: Smaller data sizes generally leads to faster searches.
|
||||
- Accelerated search process: It is based on simplified distance calculations between vectors to bitwise operations. This enables real-time querying even in extensive databases.
|
||||
|
||||
### Experiment Setup: OpenAI Embeddings in Focus
|
||||
### Experiment setup: OpenAI embeddings in focus
|
||||
|
||||
To identify Binary Quantization's impact on search efficiency and accuracy, we designed our experiment on OpenAI text-embedding models. These models, which capture nuanced linguistic features and semantic relationships, are the backbone of our analysis. We then delve deep into the potential enhancements offered by Qdrant's Binary Quantization feature.
|
||||
|
||||
@@ -74,7 +74,7 @@ This approach not only leverages the high-caliber OpenAI embeddings but also pro
|
||||
|
||||
The research employs 100K random samples from the [OpenAI 1M](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) 1M dataset, focusing on 100 randomly selected records. These records serve as queries in the experiment, aiming to assess how Binary Quantization influences search efficiency and precision within the dataset. We then use the embeddings of the queries to search for the nearest neighbors in the dataset.
|
||||
|
||||
#### Parameters: Oversampling, Rescoring, and Search Limits
|
||||
#### Parameters: oversampling, rescoring, and search limits
|
||||
|
||||
For each record, we run a parameter sweep over the number of oversampling, rescoring, and search limits. We can then understand the impact of these parameters on search accuracy and efficiency. Our experiment was designed to assess the impact of Binary Quantization under various conditions, based on the following parameters:
|
||||
|
||||
@@ -86,7 +86,7 @@ For each record, we run a parameter sweep over the number of oversampling, resco
|
||||
|
||||
Through this detailed setup, our experiment sought to shed light on the nuanced interplay between Binary Quantization and the high-quality embeddings produced by OpenAI's models. By meticulously adjusting and observing the outcomes under different conditions, we aimed to uncover actionable insights that could empower users to harness the full potential of Qdrant in combination with OpenAI's embeddings, regardless of their specific application needs.
|
||||
|
||||
### Results: Binary Quantization's Impact on OpenAI Embeddings
|
||||
### Results: binary quantization's impact on OpenAI embeddings
|
||||
|
||||
To analyze the impact of rescoring (`True` or `False`), we compared results across different model configurations and search limits. Rescoring sets up a more precise search, based on results from an initial query.
|
||||
|
||||
@@ -112,7 +112,7 @@ In contrast, for lower dimension models (such as text-embedding-3-small with 512
|
||||
|
||||
In summary, enabling rescoring dramatically improves search accuracy across all tested configurations. It is crucial feature for applications where precision is paramount. The consistent performance boost provided by rescoring underscores its value in refining search results, particularly when working with complex, high-dimensional data like OpenAI embeddings. This enhancement is critical for applications that demand high accuracy, such as semantic search, content discovery, and recommendation systems, where the quality of search results directly impacts user experience and satisfaction.
|
||||
|
||||
### Dataset Combinations
|
||||
### Dataset combinations
|
||||
|
||||
For those exploring the integration of text embedding models with Qdrant, it's crucial to consider various model configurations for optimal performance. The dataset combinations defined above illustrate different configurations to test against Qdrant. These combinations vary by two primary attributes:
|
||||
|
||||
@@ -151,7 +151,7 @@ dataset_combinations = [
|
||||
},
|
||||
]
|
||||
```
|
||||
#### Exploring Dataset Combinations and Their Impacts on Model Performance
|
||||
#### Exploring dataset combinations and their impacts on model performance
|
||||
|
||||
The code snippet iterates through predefined dataset and model combinations. For each combination, characterized by the model name and its dimensions, the corresponding experiment's results are loaded. These results, which are stored in JSON format, include performance metrics like accuracy under different configurations: with and without oversampling, and with and without a rescore step.
|
||||
|
||||
@@ -187,7 +187,7 @@ Here is a selected slice of these results, with `rescore=True`:
|
||||
|OpenAI text-embedding-3-small|1536|[DBpedia 100K](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-small-1536-100K)| 0.9847|3x|
|
||||
|OpenAI text-embedding-3-large|1536|[DBpedia 1M](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-large-1536-1M)| 0.9826|3x|
|
||||
|
||||
#### Impact of Oversampling
|
||||
#### Impact of oversampling
|
||||
|
||||
You can use oversampling in machine learning to counteract imbalances in datasets.
|
||||
It works well when one class significantly outnumbers others. This imbalance
|
||||
@@ -201,7 +201,7 @@ Without an explicit code snippet or output, we focus on the role of oversampling
|
||||
|
||||

|
||||
|
||||
### Leveraging Binary Quantization: Best Practices
|
||||
### Leveraging binary quantization: best practices
|
||||
|
||||
We recommend the following best practices for leveraging Binary Quantization to enhance OpenAI embeddings:
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ keywords:
|
||||
- memory optimization
|
||||
---
|
||||
|
||||
#### Optimizing high-dimensional vectors
|
||||
# Optimizing High-Dimensional Vectors with Binary Quantization
|
||||
|
||||
Qdrant is built to handle typical scaling challenges: high throughput, low latency and efficient indexing. **Binary quantization (BQ)** is our latest attempt to give our customers the edge they need to scale efficiently. This feature is particularly excellent for collections with large vector lengths and a large number of points.
|
||||
|
||||
@@ -44,7 +44,7 @@ For example, The 1536 dimension OpenAI embedding is worse than Open Source count
|
||||
|
||||
Our implementation of quantization achieves a good balance between full, large vectors at ranking time and binary vectors at search and retrieval time. It also has the ability for you to adjust this balance depending on your use case.
|
||||
|
||||
## Fast Search and Retrieval
|
||||
## Faster search and retrieval
|
||||
|
||||
Unlike product quantization, binary quantization does not rely on reducing the search space for each probe. Instead, we build a binary index that helps us achieve large increases in search speed.
|
||||
|
||||
@@ -54,7 +54,7 @@ HNSW is the approximate nearest neighbor search. This means our accuracy improve
|
||||
|
||||
For example, if `oversampling=2.0` and the `limit=100`, then 200 vectors will first be selected using a quantized index. For those 200 vectors, the full 32 bit vector will be used with their HNSW index to a much more accurate 100 item result set. As opposed to doing a full HNSW search, we oversample a preliminary search and then only do the full search on this much smaller set of vectors.
|
||||
|
||||
## Improved Storage Efficiency
|
||||
## Improved storage efficiency
|
||||
|
||||
The following diagram shows the binarization function, whereby we reduce 32 bits storage to 1 bit information.
|
||||
|
||||
@@ -72,13 +72,13 @@ For 100K OpenAI Embedding (`ada-002`) vectors we would need 900 Megabytes of RAM
|
||||
|
||||
This reduction in RAM needed is achieved through the compression that happens in the binary conversion. Instead of putting the HNSW index for the full vectors into RAM, we just put the binary vectors into RAM, use them for the initial oversampled search, and then use the HNSW full index of the oversampled results for the final precise search. All of this happens under the hoods without any intervention needed on your part.
|
||||
|
||||
#### When should you not use BQ?
|
||||
### When should you not use BQ?
|
||||
|
||||
Since this method exploits the over-parameterization of embedding, you can expect poorer results for small embeddings i.e. less than 1024 dimensions. With the smaller number of elements, there is not enough information maintained in the binary vector to achieve good results.
|
||||
|
||||
You will still get faster boolean operations and reduced RAM usage, but the accuracy degradation might be too high.
|
||||
|
||||
## Sample Implementation
|
||||
## Sample implementation
|
||||
|
||||
Now that we have introduced you to binary quantization, let’s try our a basic implementation. In this example, we will be using OpenAI and Cohere with Qdrant.
|
||||
|
||||
|
||||
@@ -0,0 +1,385 @@
|
||||
---
|
||||
title: "BM42: New Baseline for Hybrid Search"
|
||||
short_description: "Introducing next evolutionary step in lexical search."
|
||||
description: "Introducing BM42 - a new sparse embedding approach, which combines the benefits of exact keyword search with the intelligence of transformers."
|
||||
social_preview_image: /articles_data/bm42/social-preview.jpg
|
||||
preview_dir: /articles_data/bm42/preview
|
||||
weight: -140
|
||||
author: Andrey Vasnetsov
|
||||
date: 2024-07-01T12:00:00+03:00
|
||||
draft: false
|
||||
keywords:
|
||||
- hybrid search
|
||||
- sparse embeddings
|
||||
- bm25
|
||||
---
|
||||
|
||||
<aside role="status">
|
||||
Please note that the benchmark section of this article was updated after the publication due to a mistake in the evaluation script.
|
||||
BM42 does not outperform BM25 implementation of other vendors.
|
||||
Please consider BM42 as an experimental approach, which requires further research and development before it can be used in production.
|
||||
</aside>
|
||||
|
||||
|
||||
For the last 40 years, BM25 has served as the standard for search engines.
|
||||
It is a simple yet powerful algorithm that has been used by many search engines, including Google, Bing, and Yahoo.
|
||||
|
||||
Though it seemed that the advent of vector search would diminish its influence, it did so only partially.
|
||||
The current state-of-the-art approach to retrieval nowadays tries to incorporate BM25 along with embeddings into a hybrid search system.
|
||||
|
||||
However, the use case of text retrieval has significantly shifted since the introduction of RAG.
|
||||
Many assumptions upon which BM25 was built are no longer valid.
|
||||
|
||||
For example, the typical length of documents and queries vary significantly between traditional web search and modern RAG systems.
|
||||
|
||||
In this article, we will recap what made BM25 relevant for so long and why alternatives have struggled to replace it. Finally, we will discuss BM42, as the next step in the evolution of lexical search.
|
||||
|
||||
## Why has BM25 stayed relevant for so long?
|
||||
|
||||
To understand why, we need to analyze its components.
|
||||
|
||||
The famous BM25 formula is defined as:
|
||||
|
||||
$$
|
||||
\text{score}(D,Q) = \sum_{i=1}^{N} \text{IDF}(q_i) \times \frac{f(q_i, D) \cdot (k_1 + 1)}{f(q_i, D) + k_1 \cdot \left(1 - b + b \cdot \frac{|D|}{\text{avgdl}}\right)}
|
||||
$$
|
||||
|
||||
Let's simplify this to gain a better understanding.
|
||||
|
||||
- The $score(D, Q)$ - means that we compute the score for each pair of document $D$ and query $Q$.
|
||||
|
||||
- The $\sum_{i=1}^{N}$ - means that each of $N$ terms in the query contribute to the final score as a part of the sum.
|
||||
|
||||
- The $\text{IDF}(q_i)$ - is the inverse document frequency. The more rare the term $q_i$ is, the more it contributes to the score. A simplified formula for this is:
|
||||
|
||||
$$
|
||||
\text{IDF}(q_i) = \frac{\text{Number of documents}}{\text{Number of documents with } q_i}
|
||||
$$
|
||||
|
||||
It is fair to say that the `IDF` is the most important part of the BM25 formula.
|
||||
`IDF` selects the most important terms in the query relative to the specific document collection.
|
||||
So intuitively, we can interpret the `IDF` as **term importance within the corpora**.
|
||||
|
||||
That explains why BM25 is so good at handling queries, which dense embeddings consider out-of-domain.
|
||||
|
||||
The last component of the formula can be intuitively interpreted as **term importance within the document**.
|
||||
This might look a bit complicated, so let's break it down.
|
||||
|
||||
$$
|
||||
\text{Term importance in document }(q_i) = \color{red}\frac{f(q_i, D)\color{black} \cdot \color{blue}(k_1 + 1) \color{black} }{\color{red}f(q_i, D)\color{black} + \color{blue}k_1\color{black} \cdot \left(1 - \color{blue}b\color{black} + \color{blue}b\color{black} \cdot \frac{|D|}{\text{avgdl}}\right)}
|
||||
$$
|
||||
|
||||
- The $\color{red}f(q_i, D)\color{black}$ - is the frequency of the term $q_i$ in the document $D$. Or in other words, the number of times the term $q_i$ appears in the document $D$.
|
||||
- The $\color{blue}k_1\color{black}$ and $\color{blue}b\color{black}$ are the hyperparameters of the BM25 formula. In most implementations, they are constants set to $k_1=1.5$ and $b=0.75$. Those constants define relative implications of the term frequency and the document length in the formula.
|
||||
- The $\frac{|D|}{\text{avgdl}}$ - is the relative length of the document $D$ compared to the average document length in the corpora. The intuition befind this part is following: if the token is found in the smaller document, it is more likely that this token is important for this document.
|
||||
|
||||
#### Will BM25 term importance in the document work for RAG?
|
||||
|
||||
As we can see, the *term importance in the document* heavily depends on the statistics within the document. Moreover, statistics works well if the document is long enough.
|
||||
Therefore, it is suitable for searching webpages, books, articles, etc.
|
||||
|
||||
However, would it work as well for modern search applications, such as RAG? Let's see.
|
||||
|
||||
The typical length of a document in RAG is much shorter than that of web search. In fact, even if we are working with webpages and articles, we would prefer to split them into chunks so that
|
||||
a) Dense models can handle them and
|
||||
b) We can pinpoint the exact part of the document which is relevant to the query
|
||||
|
||||
As a result, the document size in RAG is small and fixed.
|
||||
|
||||
That effectively renders the term importance in the document part of the BM25 formula useless.
|
||||
The term frequency in the document is always 0 or 1, and the relative length of the document is always 1.
|
||||
|
||||
So, the only part of the BM25 formula that is still relevant for RAG is `IDF`. Let's see how we can leverage it.
|
||||
|
||||
## Why SPLADE is not always the answer
|
||||
|
||||
Before discussing our new approach, let's examine the current state-of-the-art alternative to BM25 - SPLADE.
|
||||
|
||||
The idea behind SPLADE is interesting—what if we let a smart, end-to-end trained model generate a bag-of-words representation of the text for us?
|
||||
It will assign all the weights to the tokens, so we won't need to bother with statistics and hyperparameters.
|
||||
The documents are then represented as a sparse embedding, where each token is represented as an element of the sparse vector.
|
||||
|
||||
And it works in academic benchmarks. Many papers report that SPLADE outperforms BM25 in terms of retrieval quality.
|
||||
This performance, however, comes at a cost.
|
||||
|
||||
* **Inappropriate Tokenizer**: To incorporate transformers for this task, SPLADE models require using a standard transformer tokenizer. These tokenizers are not designed for retrieval tasks. For example, if the word is not in the (quite limited) vocabulary, it will be either split into subwords or replaced with a `[UNK]` token. This behavior works well for language modeling but is completely destructive for retrieval tasks.
|
||||
|
||||
* **Expensive Token Expansion**: In order to compensate the tokenization issues, SPLADE uses *token expansion* technique. This means that we generate a set of similar tokens for each token in the query. There are a few problems with this approach:
|
||||
- It is computationally and memory expensive. We need to generate more values for each token in the document, which increases both the storage size and retrieval time.
|
||||
- It is not always clear where to stop with the token expansion. The more tokens we generate, the more likely we are to get the relevant one. But simultaneously, the more tokens we generate, the more likely we are to get irrelevant results.
|
||||
- Token expansion dilutes the interpretability of the search. We can't say which tokens were used in the document and which were generated by the token expansion.
|
||||
|
||||
* **Domain and Language Dependency**: SPLADE models are trained on specific corpora. This means that they are not always generalizable to new or rare domains. As they don't use any statistics from the corpora, they cannot adapt to the new domain without fine-tuning.
|
||||
|
||||
* **Inference Time**: Additionally, currently available SPLADE models are quite big and slow. They usually require a GPU to make the inference in a reasonable time.
|
||||
|
||||
At Qdrant, we acknowledge the aforementioned problems and are looking for a solution.
|
||||
Our idea was to combine the best of both worlds - the simplicity and interpretability of BM25 and the intelligence of transformers while avoiding the pitfalls of SPLADE.
|
||||
|
||||
And here is what we came up with.
|
||||
|
||||
## The best of both worlds
|
||||
|
||||
As previously mentioned, `IDF` is the most important part of the BM25 formula. In fact it is so important, that we decided to build its calculation into the Qdrant engine itself.
|
||||
Check out our latest [release notes](https://github.com/qdrant/qdrant/releases/tag/v1.10.0). This type of separation allows streaming updates of the sparse embeddings while keeping the `IDF` calculation up-to-date.
|
||||
|
||||
As for the second part of the formula, *the term importance within the document* needs to be rethought.
|
||||
|
||||
Since we can't rely on the statistics within the document, we can try to use the semantics of the document instead.
|
||||
And semantics is what transformers are good at. Therefore, we only need to solve two problems:
|
||||
|
||||
- How does one extract the importance information from the transformer?
|
||||
- How can tokenization issues be avoided?
|
||||
|
||||
|
||||
### Attention is all you need
|
||||
|
||||
Transformer models, even those used to generate embeddings, generate a bunch of different outputs.
|
||||
Some of those outputs are used to generate embeddings.
|
||||
|
||||
Others are used to solve other kinds of tasks, such as classification, text generation, etc.
|
||||
|
||||
The one particularly interesting output for us is the attention matrix.
|
||||
|
||||
{{< figure src="/articles_data/bm42/attention-matrix.png" alt="Attention matrix" caption="Attention matrix" width="60%" >}}
|
||||
|
||||
The attention matrix is a square matrix, where each row and column corresponds to the token in the input sequence.
|
||||
It represents the importance of each token in the input sequence for each other.
|
||||
|
||||
The classical transformer models are trained to predict masked tokens in the context, so the attention weights define which context tokens influence the masked token most.
|
||||
|
||||
Apart from regular text tokens, the transformer model also has a special token called `[CLS]`. This token represents the whole sequence in the classification tasks, which is exactly what we need.
|
||||
|
||||
By looking at the attention row for the `[CLS]` token, we can get the importance of each token in the document for the whole document.
|
||||
|
||||
|
||||
```python
|
||||
sentences = "Hello, World - is the starting point in most programming languages"
|
||||
|
||||
features = transformer.tokenize(sentences)
|
||||
|
||||
# ...
|
||||
|
||||
attentions = transformer.auto_model(**features, output_attentions=True).attentions
|
||||
|
||||
weights = torch.mean(attentions[-1][0,:,0], axis=0)
|
||||
# ▲ ▲ ▲ ▲
|
||||
# │ │ │ └─── [CLS] token is the first one
|
||||
# │ │ └─────── First item of the batch
|
||||
# │ └────────── Last transformer layer
|
||||
# └────────────────────────── Averate all 6 attention heads
|
||||
|
||||
for weight, token in zip(weights, tokens):
|
||||
print(f"{token}: {weight}")
|
||||
|
||||
# [CLS] : 0.434 // Filter out the [CLS] token
|
||||
# hello : 0.039
|
||||
# , : 0.039
|
||||
# world : 0.107 // <-- The most important token
|
||||
# - : 0.033
|
||||
# is : 0.024
|
||||
# the : 0.031
|
||||
# starting : 0.054
|
||||
# point : 0.028
|
||||
# in : 0.018
|
||||
# most : 0.016
|
||||
# programming : 0.060 // <-- The third most important token
|
||||
# languages : 0.062 // <-- The second most important token
|
||||
# [SEP] : 0.047 // Filter out the [SEP] token
|
||||
|
||||
```
|
||||
|
||||
|
||||
The resulting formula for the BM42 score would look like this:
|
||||
|
||||
$$
|
||||
\text{score}(D,Q) = \sum_{i=1}^{N} \text{IDF}(q_i) \times \text{Attention}(\text{CLS}, q_i)
|
||||
$$
|
||||
|
||||
|
||||
Note that classical transformers have multiple attention heads, so we can get multiple importance vectors for the same document. The simplest way to combine them is to simply average them.
|
||||
|
||||
These averaged attention vectors make up the importance information we were looking for.
|
||||
The best part is, one can get them from any transformer model, without any additional training.
|
||||
Therefore, BM42 can support any natural language as long as there is a transformer model for it.
|
||||
|
||||
In our implementation, we use the `sentence-transformers/all-MiniLM-L6-v2` model, which gives a huge boost in the inference speed compared to the SPLADE models. In practice, any transformer model can be used.
|
||||
It doesn't require any additional training, and can be easily adapted to work as BM42 backend.
|
||||
|
||||
|
||||
### WordPiece retokenization
|
||||
|
||||
The final piece of the puzzle we need to solve is the tokenization issue. In order to get attention vectors, we need to use native transformer tokenization.
|
||||
But this tokenization is not suitable for the retrieval tasks. What can we do about it?
|
||||
|
||||
Actually, the solution we came up with is quite simple. We reverse the tokenization process after we get the attention vectors.
|
||||
|
||||
Transformers use [WordPiece](https://huggingface.co/learn/nlp-course/en/chapter6/6) tokenization.
|
||||
In case it sees the word, which is not in the vocabulary, it splits it into subwords.
|
||||
|
||||
Here is how that looks:
|
||||
|
||||
```text
|
||||
"unbelievable" -> ["un", "##believ", "##able"]
|
||||
```
|
||||
|
||||
What can merge the subwords back into the words. Luckily, the subwords are marked with the `##` prefix, so we can easily detect them.
|
||||
Since the attention weights are normalized, we can simply sum the attention weights of the subwords to get the attention weight of the word.
|
||||
|
||||
After that, we can apply the same traditional NLP techniques, as
|
||||
|
||||
- Removing of the stop-words
|
||||
- Removing of the punctuation
|
||||
- Lemmatization
|
||||
|
||||
In this way, we can significantly reduce the number of tokens, and therefore minimize the memory footprint of the sparse embeddings. We won't simultaneously compromise the ability to match (almost) exact tokens.
|
||||
|
||||
## Practical examples
|
||||
|
||||
|
||||
| Trait | BM25 | SPLADE | BM42 |
|
||||
|-------------------------|--------------|--------------|--------------|
|
||||
| Interpretability | High ✅ | Ok 🆗 | High ✅ |
|
||||
| Document Inference speed| Very high ✅ | Slow 🐌 | High ✅ |
|
||||
| Query Inference speed | Very high ✅ | Slow 🐌 | Very high ✅ |
|
||||
| Memory footprint | Low ✅ | High ❌ | Low ✅ |
|
||||
| In-domain accuracy | Ok 🆗 | High ✅ | High ✅ |
|
||||
| Out-of-domain accuracy | Ok 🆗 | Low ❌ | Ok 🆗 |
|
||||
| Small documents accuracy| Low ❌ | High ✅ | High ✅ |
|
||||
| Large documents accuracy| High ✅ | Low ❌ | Ok 🆗 |
|
||||
| Unknown tokens handling | Yes ✅ | Bad ❌ | Yes ✅ |
|
||||
| Multi-lingual support | Yes ✅ | No ❌ | Yes ✅ |
|
||||
| Best Match | Yes ✅ | No ❌ | Yes ✅ |
|
||||
|
||||
|
||||
Starting from Qdrant v1.10.0, BM42 can be used in Qdrant via FastEmbed inference.
|
||||
|
||||
Let's see how you can setup a collection for hybrid search with BM42 and [jina.ai](https://jina.ai/embeddings/) dense embeddings.
|
||||
|
||||
```http
|
||||
PUT collections/my-hybrid-collection
|
||||
{
|
||||
"vectors": {
|
||||
"jina": {
|
||||
"size": 768,
|
||||
"distance": "Cosine"
|
||||
}
|
||||
},
|
||||
"sparse_vectors": {
|
||||
"bm42": {
|
||||
"modifier": "idf" // <--- This parameter enables the IDF calculation
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
|
||||
client = QdrantClient()
|
||||
|
||||
client.create_collection(
|
||||
collection_name="my-hybrid-collection",
|
||||
vectors_config={
|
||||
"jina": models.VectorParams(
|
||||
size=768,
|
||||
distance=models.Distance.COSINE,
|
||||
)
|
||||
},
|
||||
sparse_vectors_config={
|
||||
"bm42": models.SparseVectorParams(
|
||||
modifier=models.Modifier.IDF,
|
||||
)
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
The search query will retrieve the documents with both dense and sparse embeddings and combine the scores
|
||||
using the Reciprocal Rank Fusion (RRF) algorithm.
|
||||
|
||||
```python
|
||||
from fastembed import SparseTextEmbedding, TextEmbedding
|
||||
|
||||
query_text = "best programming language for beginners?"
|
||||
|
||||
model_bm42 = SparseTextEmbedding(model_name="Qdrant/bm42-all-minilm-l6-v2-attentions")
|
||||
model_jina = TextEmbedding(model_name="jinaai/jina-embeddings-v2-base-en")
|
||||
|
||||
sparse_embedding = list(embedding_model.query_embed(query_text))[0]
|
||||
dense_embedding = list(model_jina.query_embed(query_text))[0]
|
||||
|
||||
client.query_points(
|
||||
collection_name="my-hybrid-collection",
|
||||
prefetch=[
|
||||
models.Prefetch(query=sparse_embedding.as_object(), using="bm42", limit=10),
|
||||
models.Prefetch(query=dense_embedding.tolist(), using="jina", limit=10),
|
||||
],
|
||||
query=models.FusionQuery(fusion=models.Fusion.RRF), # <--- Combine the scores
|
||||
limit=10
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
### Benchmarks
|
||||
|
||||
To prove the point further we have conducted some benchmarks to highlight the cases where BM42 outperforms BM25.
|
||||
Please note, that we didn't intend to make an exhaustive evaluation, as we are presenting a new approach, not a new model.
|
||||
|
||||
For out experiments we choose [quora](https://huggingface.co/datasets/BeIR/quora) dataset, which represents a question-deduplication task ~~the Question-Answering task~~.
|
||||
|
||||
|
||||
The typical example of the dataset is the following:
|
||||
|
||||
```text
|
||||
{"_id": "109", "text": "How GST affects the CAs and tax officers?"}
|
||||
{"_id": "110", "text": "Why can't I do my homework?"}
|
||||
{"_id": "111", "text": "How difficult is it get into RSI?"}
|
||||
```
|
||||
|
||||
As you can see, it has pretty short texts, there are not much of the statistics to rely on.
|
||||
|
||||
After encoding with BM42, the average vector size is only **5.6 elements per document**.
|
||||
|
||||
With `datatype: uint8` available in Qdrant, the total size of the sparse vector index is about **13Mb** for ~530k documents.
|
||||
|
||||
As a reference point, we use:
|
||||
|
||||
- BM25 with tantivy
|
||||
- the [sparse vector BM25 implementation](https://github.com/qdrant/bm42_eval/blob/master/index_bm25_qdrant.py) with the same preprocessing pipeline like for BM42: tokenization, stop-words removal, and lemmatization
|
||||
|
||||
| | BM25 (tantivy) | BM25 (Sparse) | BM42 |
|
||||
|----------------------|-------------------|---------------|----------|
|
||||
| ~~Precision @ 10~~ * | ~~0.45~~ | ~~0.45~~ | ~~0.49~~ |
|
||||
| Recall @ 10 | ~~0.71~~ **0.89** | 0.83 | 0.85 |
|
||||
|
||||
|
||||
\* - values were corrected after the publication due to a mistake in the evaluation script.
|
||||
|
||||
<aside role="status">
|
||||
When used properly, BM25 with tantivy achieves the best results. Our initial implementation performed wrong character escaping that led to understating the value of <code>recall@10</code> for tantivy.
|
||||
</aside>
|
||||
|
||||
To make our benchmarks transparent, we have published scripts we used for the evaluation: see [github repo](https://github.com/qdrant/bm42_eval).
|
||||
|
||||
|
||||
Please note, that both BM25 and BM42 won't work well on their own in a production environment.
|
||||
Best results are achieved with a combination of sparse and dense embeddings in a hybrid approach.
|
||||
In this scenario, the two models are complementary to each other.
|
||||
The sparse model is responsible for exact token matching, while the dense model is responsible for semantic matching.
|
||||
|
||||
Some more advanced models might outperform default `sentence-transformers/all-MiniLM-L6-v2` model we were using.
|
||||
We encourage developers involved in training embedding models to include a way to extract attention weights and contribute to the BM42 backend.
|
||||
|
||||
## Fostering curiosity and experimentation
|
||||
|
||||
Despite all of its advantages, BM42 is not always a silver bullet.
|
||||
For large documents without chunks, BM25 might still be a better choice.
|
||||
|
||||
There might be a smarter way to extract the importance information from the transformer. There could be a better method to weigh IDF against attention scores.
|
||||
|
||||
Qdrant does not specialize in model training. Our core project is the search engine itself. However, we understand that we are not operating in a vacuum. By introducing BM42, we are stepping up to empower our community with novel tools for experimentation.
|
||||
|
||||
We truly believe that the sparse vectors method is at exact level of abstraction to yield both powerful and flexible results.
|
||||
|
||||
Many of you are sharing your recent Qdrant projects in our [Discord channel](https://discord.com/invite/qdrant). Feel free to try out BM42 and let us know what you come up with.
|
||||
|
||||
@@ -0,0 +1,249 @@
|
||||
---
|
||||
title: " Data Privacy with Qdrant: Implementing Role-Based Access Control (RBAC)" #required
|
||||
short_description: "Secure Your Data with Qdrant: Implementing RBAC"
|
||||
description: Discover how Qdrant's Role-Based Access Control (RBAC) ensures data privacy and compliance for your AI applications. Build secure and scalable systems with ease. Read more now!
|
||||
social_preview_image: /articles_data/data-privacy/preview/social_preview.jpg # This image will be used in social media previews, should be 1200x630px. Required.
|
||||
preview_dir: /articles_data/data-privacy/preview # This directory contains images that will be used in the article preview. They can be generated from one image. Read more below. Required.
|
||||
weight: -110 # This is the order of the article in the list of articles at the footer. The lower the number, the higher the article will be in the list.
|
||||
author: Qdrant Team # Author of the article. Required.
|
||||
author_link: https://qdrant.tech/ # Link to the author's page. Required.
|
||||
date: 2024-06-18T08:00:00-03:00 # Date of the article. Required.
|
||||
draft: false # If true, the article will not be published
|
||||
keywords: # Keywords for SEO
|
||||
- Role-Based Access Control (RBAC)
|
||||
- Data Privacy in Vector Databases
|
||||
- Secure AI Data Management
|
||||
- Qdrant Data Security
|
||||
- Enterprise Data Compliance
|
||||
---
|
||||
|
||||
Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures become critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as "[embedding inversion attack](https://arxiv.org/abs/2004.00053)," where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves.
|
||||
|
||||
Strict compliance regulations govern data stored in vector databases across various industries. For instance, healthcare must comply with HIPAA, which dictates how protected health information (PHI) is stored, transmitted, and secured. Similarly, the financial services industry follows PCI DSS to safeguard sensitive financial data. These regulations require developers to ensure data storage and transmission comply with industry-specific legal frameworks across different regions. **As a result, features that enable data privacy, security and sovereignty are deciding factors when choosing the right vector database.**
|
||||
|
||||
This article explores various strategies to ensure the security of your critical data while leveraging the benefits of vector search. Implementing some of these security approaches can help you build privacy-enhanced similarity search algorithms and integrate them into your AI applications.
|
||||
Additionally, you will learn how to build a fully data-sovereign architecture, allowing you to retain control over your data and comply with relevant data laws and regulations.
|
||||
|
||||
> To skip right to the code implementation, [click here](/articles/data-privacy/#jwt-on-qdrant).
|
||||
|
||||
## Vector Database Security: An Overview
|
||||
|
||||
Vector databases are often unsecured by default to facilitate rapid prototyping and experimentation. This approach allows developers to quickly ingest data, build vector representations, and test similarity search algorithms without initial security concerns. However, in production environments, unsecured databases pose significant data breach risks.
|
||||
|
||||
For production use, robust security systems are essential. Authentication, particularly using static API keys, is a common approach to control access and prevent unauthorized modifications. Yet, simple API authentication is insufficient for enterprise data, which requires granular control.
|
||||
|
||||
The primary challenge with static API keys is their all-or-nothing access, inadequate for role-based data segregation in enterprise applications. Additionally, a compromised key could grant attackers full access to manipulate or steal data. To strengthen the security of the vector database, developers typically need the following:
|
||||
|
||||
1. **Encryption**: This ensures that sensitive data is scrambled as it travels between the application and the vector database. This safeguards against Man-in-the-Middle ([MitM](https://en.wikipedia.org/wiki/Man-in-the-middle_attack)) attacks, where malicious actors can attempt to intercept and steal data during transmission.
|
||||
2. **Role-Based Access Control**: As mentioned before, traditional static API keys grant all-or-nothing access, which is a significant security risk in enterprise environments. RBAC offers a more granular approach by defining user roles and assigning specific data access permissions based on those roles. For example, an analyst might have read-only access to specific datasets, while an administrator might have full CRUD (Create, Read, Update, Delete) permissions across the database.
|
||||
3. **Deployment Flexibility**: Data residency regulations like GDPR (General Data Protection Regulation) and industry-specific compliance requirements dictate where data can be stored, processed, and accessed. Developers would need to choose a database solution which offers deployment options that comply with these regulations. This might include on-premise deployments within a company's private cloud or geographically distributed cloud deployments that adhere to data residency laws.
|
||||
|
||||
## How Qdrant Handles Data Privacy and Security
|
||||
|
||||
One of the cornerstones of our design choices at Qdrant has been the focus on security features. We have built in a range of features keeping the enterprise user in mind, which allow building of granular access control on a fully data sovereign architecture.
|
||||
|
||||
A Qdrant instance is unsecured by default. However, when you are ready to deploy in production, Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture.
|
||||
|
||||
### API Keys and TLS Encryption
|
||||
|
||||
For simpler use cases, Qdrant offers API key-based authentication. This includes both regular API keys and read-only API keys. Regular API keys grant full access to read, write, and delete operations, while read-only keys restrict access to data retrieval operations only, preventing write actions.
|
||||
|
||||
On Qdrant Cloud, you can create API keys using the [Cloud Dashboard](https://qdrant.to/cloud). This allows you to generate API keys that give you access to a single node or cluster, or multiple clusters. You can read the steps to do so [here](/documentation/cloud/authentication/).
|
||||
|
||||

|
||||
|
||||
For on-premise or local deployments, you'll need to configure API key authentication. This involves specifying a key in either the Qdrant configuration file or as an environment variable. This ensures that all requests to the server must include a valid API key sent in the header.
|
||||
|
||||
When using the simple API key-based authentication, you should also turn on TLS encryption. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](/documentation/guides/security/#tls) in the configuration.
|
||||
|
||||
API authentication, coupled with TLS encryption, offers a first layer of security for your Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs).
|
||||
|
||||
### JWT on Qdrant
|
||||
|
||||
JSON Web Tokens (JWTs) are a compact, URL-safe, and stateless means of representing _claims_ to be transferred between two parties. These claims are encoded as a JSON object and are cryptographically signed.
|
||||
|
||||
JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity.
|
||||
|
||||
In Qdrant, JWT forms the foundation through which powerful access controls can be built. Let’s understand how.
|
||||
|
||||
JWT is enabled on the Qdrant instance by specifying the API key and turning on the **jwt_rbac** feature in the configuration (alternatively, they can be set as environment variables). For any subsequent request, the API key is used to encode or decode the token.
|
||||
|
||||
The way JWT works is that just the API key is enough to generate the token, and doesn’t require any communication with the Qdrant instance or server. There are several libraries that help generate tokens by encoding a payload, such as [PyJWT](https://pyjwt.readthedocs.io/en/stable/) (for Python), [jsonwebtoken](https://www.npmjs.com/package/jsonwebtoken) (for JavaScript), and [jsonwebtoken](https://crates.io/crates/jsonwebtoken) (for Rust). Qdrant uses the HS256 algorithm to encode or decode the tokens.
|
||||
|
||||
We will look at the payload structure shortly, but here’s how you can generate a token using PyJWT.
|
||||
|
||||
```python
|
||||
import jwt
|
||||
import datetime
|
||||
|
||||
# Define your API key and other payload data
|
||||
api_key = "your_api_key"
|
||||
payload = { ...
|
||||
}
|
||||
|
||||
token = jwt.encode(payload, api_key, algorithm="HS256")
|
||||
print(token)
|
||||
```
|
||||
|
||||
Once you have generated the token, you should include it in the subsequent requests. You can do so by providing it as a bearer token in the Authorization header, or in the API Key header of your requests.
|
||||
|
||||
Below is an example of how to do so using QdrantClient in Python:
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
qdrant_client = QdrantClient(
|
||||
"http://localhost:6333",
|
||||
api_key="<JWT>", # the token goes here
|
||||
)
|
||||
# Example search vector
|
||||
search_vector = [0.1, 0.2, 0.3, 0.4]
|
||||
|
||||
# Example similarity search request
|
||||
response = qdrant_client.search(
|
||||
collection_name="demo_collection",
|
||||
query_vector=search_vector,
|
||||
limit=5 # Number of results to retrieve
|
||||
)
|
||||
```
|
||||
|
||||
For convenience, we have added a JWT generation tool in the Qdrant Web UI, which is present under the 🔑 tab. For your local deployments, you will find it at [http://localhost:6333/dashboard#/jwt](http://localhost:6333/dashboard#/jwt).
|
||||
|
||||
### Payload Configuration
|
||||
|
||||
There are several different options (claims) you can use in the JWT payload that help control access and functionality. Let’s look at them one by one.
|
||||
|
||||
**exp**: This claim is the expiration time of the token, and is a unix timestamp in seconds. After the expiration time, the token will be invalid.
|
||||
|
||||
**value_exists**: This claim validates the token against a specific key-value stored in a collection. By using this claim, you can revoke access by simply changing a value without having to invalidate the API key.
|
||||
|
||||
**access**: This claim defines the access level of the token. The access level can be global read (r) or manage (m). It can also be specific to a collection, or even a subset of a collection, using read (r) and read-write (rw).
|
||||
|
||||
Let’s look at a few example JWT payload configurations.
|
||||
|
||||
**Scenario 1: 1-hour expiry time, and read-only access to a collection**
|
||||
```json
|
||||
{
|
||||
"exp": 1690995200, // Set to 1 hour from the current time (Unix timestamp)
|
||||
"access": [
|
||||
{
|
||||
"collection": "demo_collection",
|
||||
"access": "r" // Read-only access
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
```
|
||||
|
||||
**Scenario 2: 1-hour expiry time, and access to user with a specific role**
|
||||
|
||||
Suppose you have a ‘users’ collection and have defined specific roles for each user, such as ‘developer’, ‘manager’, ‘admin’, ‘analyst’, and ‘revoked’. In such a scenario, you can use a combination of **exp** and **value_exists**.
|
||||
```json
|
||||
{
|
||||
"exp": 1690995200,
|
||||
"value_exists": {
|
||||
"collection": "users",
|
||||
"matches": [
|
||||
{ "key": "username", "value": "john" },
|
||||
{ "key": "role", "value": "developer" }
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
```
|
||||
|
||||
|
||||
|
||||
Now, if you ever want to revoke access for a user, simply change the value of their role. All future requests will be invalid using a token payload of the above type.
|
||||
|
||||
**Scenario 3: 1-hour expiry time, and read-write access to a subset of a collection**
|
||||
|
||||
You can even specify access levels specific to subsets of a collection. This can be especially useful when you are leveraging [multitenancy](/documentation/guides/multiple-partitions/), and want to segregate access.
|
||||
```json
|
||||
{
|
||||
"exp": 1690995200,
|
||||
"access": [
|
||||
{
|
||||
"collection": "demo_collection",
|
||||
"access": "r",
|
||||
"payload": {
|
||||
"user_id": "user_123456"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
By combining the claims, you can fully customize the access level that a user or a role has within the vector store.
|
||||
|
||||
### Creating Role-Based Access Control (RBAC) Using JWT
|
||||
|
||||
As we saw above, JWT claims create powerful levers through which you can create granular access control on Qdrant. Let’s bring it all together and understand how it helps you create Role-Based Access Control (RBAC).
|
||||
|
||||
In a typical enterprise application, you will have a segregation of users based on their roles and permissions. These could be:
|
||||
|
||||
1. **Admin or Owner:** with full access, and can generate API keys.
|
||||
2. **Editor:** with read-write access levels to specific collections.
|
||||
3. **Viewer:** with read-only access to specific collections.
|
||||
4. **Data Scientist or Analyst:** with read-only access to specific collections.
|
||||
5. **Developer:** with read-write access to development- or testing-specific collections, but limited access to production data.
|
||||
6. **Guest:** with limited read-only access to publicly available collections.
|
||||
|
||||
In addition, you can create access levels within sections of a collection. In a multi-tenant application, where you have used payload-based partitioning, you can create read-only access for specific user roles for a subset of the collection that belongs to that user.
|
||||
|
||||
Your application requirements will eventually help you decide the roles and access levels you should create. For example, in an application managing customer data, you could create additional roles such as:
|
||||
|
||||
**Customer Support Representative**: read-write access to customer service-related data but no access to billing information.
|
||||
|
||||
**Billing Department**: read-only access to billing data and read-write access to payment records.
|
||||
|
||||
**Marketing Analyst**: read-only access to anonymized customer data for analytics.
|
||||
|
||||
Each role can be assigned a JWT with claims that specify expiration times, read/write permissions for collections, and validating conditions.
|
||||
|
||||
In such an application, an example JWT payload for a customer support representative role could be:
|
||||
|
||||
```json
|
||||
{
|
||||
"exp": 1690995200,
|
||||
"access": [
|
||||
{
|
||||
"collection": "customer_data",
|
||||
"access": "rw",
|
||||
"payload": {
|
||||
"department": "support"
|
||||
}
|
||||
}
|
||||
],
|
||||
"value_exists": {
|
||||
"collection": "departments",
|
||||
"matches": [
|
||||
{ "key": "department", "value": "support" }
|
||||
]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
As you can see, by implementing RBAC, you can ensure proper segregation of roles and their privileges, and avoid privacy loopholes in your application.
|
||||
|
||||
## Qdrant Hybrid Cloud and Data Sovereignty
|
||||
|
||||
Data governance varies by country, especially for global organizations dealing with different regulations on data privacy, security, and access. This often necessitates deploying infrastructure within specific geographical boundaries.
|
||||
|
||||
To address these needs, the vector database you choose should support deployment and scaling within your controlled infrastructure. [Qdrant Hybrid Cloud](/documentation/hybrid-cloud/) offers this flexibility, along with features like sharding, replicas, JWT authentication, and monitoring.
|
||||
|
||||
Qdrant Hybrid Cloud integrates Kubernetes clusters from various environments—cloud, on-premises, or edge—into a unified managed service. This allows organizations to manage Qdrant databases through the Qdrant Cloud UI while keeping the databases within their infrastructure.
|
||||
|
||||
With JWT and RBAC, Qdrant Hybrid Cloud provides a secure, private, and sovereign vector store. Enterprises can scale their AI applications geographically, comply with local laws, and maintain strict data control.
|
||||
|
||||
## Conclusion
|
||||
|
||||
Vector similarity is increasingly becoming the backbone of AI applications that leverage unstructured data. By transforming data into vectors – their numerical representations – organizations can build powerful applications that harness semantic search, ranging from better recommendation systems to algorithms that help with personalization, or powerful customer support chatbots.
|
||||
|
||||
However, to fully leverage the power of AI in production, organizations need to choose a vector database that offers strong privacy and security features, while also helping them adhere to local laws and regulations.
|
||||
|
||||
Qdrant provides exceptional efficiency and performance, along with the capability to implement granular access control to data, Role-Based Access Control (RBAC), and the ability to build a fully data-sovereign architecture.
|
||||
|
||||
Interested in mastering vector search security and deployment strategies? [Join our Discord community](https://discord.gg/qdrant) to explore more advanced search strategies, connect with other developers and researchers in the industry, and stay updated on the latest innovations!
|
||||
@@ -1,16 +1,16 @@
|
||||
---
|
||||
title: "Discovery needs context" #required
|
||||
short_description: Discover points by constraining the space.
|
||||
description: Qdrant released a new functionality that lets you constrain the space in which a search is performed, relying only on vectors. #required
|
||||
social_preview_image: /articles_data/discovery-search/social_preview.jpg # This image will be used in social media previews, should be 1200x630px. Required.
|
||||
small_preview_image: /articles_data/discovery-search/icon.svg # This image will be used in the list of articles at the footer, should be 40x40px
|
||||
preview_dir: /articles_data/discovery-search/preview # This directory contains images that will be used in the article preview. They can be generated from one image. Read more below. Required.
|
||||
weight: -110 # This is the order of the article in the list of articles at the footer. The lower the number, the higher the article will be in the list.
|
||||
author: Luis Cossío # Author of the article. Required.
|
||||
author_link: https://coszio.github.io # Link to the author's page. Required.
|
||||
date: 2024-01-31T08:00:00-03:00 # Date of the article. Required.
|
||||
draft: false # If true, the article will not be published
|
||||
keywords: # Keywords for SEO
|
||||
title: "Discovery Search: A New Approach to Vector Space"
|
||||
short_description: Discovery Search, an innovative API for precise, tailored search results.
|
||||
description: Explore the next frontier in search technology with Discovery Search. Learn how this innovative API provides precise and tailored results.
|
||||
social_preview_image: /articles_data/discovery-search/social_preview.jpg
|
||||
small_preview_image: /articles_data/discovery-search/icon.svg
|
||||
preview_dir: /articles_data/discovery-search/preview
|
||||
weight: -110
|
||||
author: Luis Cossío
|
||||
author_link: https://coszio.github.io
|
||||
date: 2024-01-31T08:00:00-03:00
|
||||
draft: false
|
||||
keywords:
|
||||
- why use a vector database
|
||||
- specialty
|
||||
- search
|
||||
@@ -19,14 +19,24 @@ keywords: # Keywords for SEO
|
||||
- vector-search
|
||||
---
|
||||
|
||||
# How to Master Vector Space Exploration with Discovery Search
|
||||
|
||||
When Christopher Columbus and his crew sailed to cross the Atlantic Ocean, they were not looking for America. They were looking for a new route to India, and they were convinced that the Earth was round. They didn't know anything about America, but since they were going west, they stumbled upon it.
|
||||
|
||||
They couldn't reach their _target_, because the geography didn't let them, but once they realized it wasn't India, they claimed it a new "discovery" for their crown. If we consider that sailors need water to sail, then we can establish a _context_ which is positive in the water, and negative on land. Once the sailor's search was stopped by the land, they could not go any further, and a new route was found. Let's keep this concepts of _target_ and _context_ in mind as we explore the new functionality of Qdrant: __Discovery search__.
|
||||
They couldn't reach their _target_, because the geography didn't let them, but once they realized it wasn't India, they claimed it a new "discovery" for their crown. If we consider that sailors need water to sail, then we can establish a _context_ which is positive in the water, and negative on land. Once the sailor's search was stopped by the land, they could not go any further, and a new route was found. Let's keep these concepts of _target_ and _context_ in mind as we explore the new functionality of Qdrant: __Discovery search__.
|
||||
|
||||
## What is discovery search?
|
||||
|
||||
Discovery search is a powerful tool that lets you explore the vector space in a more controlled way. It can be used to find points that are not necessarily close to the target but are still relevant to the search. It can also be used to represent complex tastes and break out of the similarity bubble. Check out the documentation to learn more about the math behind it and how to use it.
|
||||
|
||||
## Qdrant's discovery search: version 1.7 release
|
||||
|
||||
In version 1.7, Qdrant [released](/articles/qdrant-1.7.x/) this novel API that lets you constrain the space in which a search is performed, relying only on pure vectors. This is a powerful tool that lets you explore the vector space in a more controlled way. It can be used to find points that are not necessarily closest to the target, but are still relevant to the search.
|
||||
|
||||
You can already select which points are available to the search by using payload filters. This by itself is very versatile because it allows us to craft complex filters that show only the points that satisfy their criteria deterministically. However, the payload associated with each point is arbitrary and cannot tell us anything about their position in the vector space. In other words, filtering out irrelevant points can be seen as creating a _mask_ rather than a hyperplane –cutting in between the positive and negative vectors– in the space.
|
||||
|
||||
## Understanding context in discovery search
|
||||
|
||||
This is where a __vector _context___ can help. We define _context_ as a list of pairs. Each pair is made up of a positive and a negative vector. With a context, we can define hyperplanes within the vector space, which always prefer the positive over the negative vectors. This effectively partitions the space where the search is performed. After the space is partitioned, we then need a _target_ to return the points that are more similar to it.
|
||||
|
||||

|
||||
@@ -42,7 +52,7 @@ While positive and negative vectors might suggest the use of the <a href="/docum
|
||||
|
||||
However, it is not the only way to use it. Alternatively, you can __only__ provide a context, which invokes a [__Context Search__](#context-search). This is useful when you want to explore the space defined by the context, but don't have a specific target in mind. But hold your horses, we'll get to that [later ↪](#context-search).
|
||||
|
||||
## Discovery search
|
||||
## Real-world discovery search applications
|
||||
|
||||
Let's talk about the first case: context with a target.
|
||||
|
||||
@@ -61,11 +71,11 @@ Turns out, multimodal encoders <a href="https://modalitygap.readthedocs.io/en/la
|
||||
|
||||

|
||||
|
||||
This is where discovery excels, because it allows us to constrain the space considering the same mode (images) while using a target from the other mode (text).
|
||||
This is where discovery excels because it allows us to constrain the space considering the same mode (images) while using a target from the other mode (text).
|
||||
|
||||

|
||||
|
||||
Discovery also lets us keep giving feedback to the search engine in the shape of more context pairs, so we can keep refining our search until we find what we are looking for.
|
||||
Discovery search also lets us keep giving feedback to the search engine in the shape of more context pairs, so we can keep refining our search until we find what we are looking for.
|
||||
|
||||
Another intuitive example: imagine you're looking for a fish pizza, but pizza names can be confusing, so you can just type "pizza", and prefer a fish over meat. Discovery search will let you use these inputs to suggest a fish pizza... even if it's not called fish pizza!
|
||||
|
||||
@@ -73,9 +83,9 @@ Another intuitive example: imagine you're looking for a fish pizza, but pizza na
|
||||
|
||||
## Context search
|
||||
|
||||
Now, second case: only providing context.
|
||||
Now, the second case: only providing context.
|
||||
|
||||
Ever been caught in the same recommendations on your favourite music streaming service? This may be caused by getting stuck in a similarity bubble. As user input gets more complex, diversity becomes scarce, and it becomes harder to force the system to recommend something different.
|
||||
Ever been caught in the same recommendations on your favorite music streaming service? This may be caused by getting stuck in a similarity bubble. As user input gets more complex, diversity becomes scarce, and it becomes harder to force the system to recommend something different.
|
||||
|
||||

|
||||
|
||||
@@ -83,12 +93,14 @@ __Context search__ solves this by de-focusing the search around a single point.
|
||||
|
||||

|
||||
|
||||
Creating complex tastes in a high-dimensional space becomes easier, since you can just add more context pairs to the search. This way, you should be able to constrain the space enough so you select points from a per-search "category" created just from the context in the input.
|
||||
Creating complex tastes in a high-dimensional space becomes easier since you can just add more context pairs to the search. This way, you should be able to constrain the space enough so you select points from a per-search "category" created just from the context in the input.
|
||||
|
||||

|
||||
|
||||
This way you can give refeshing recommendations, while still being in control by providing positive and negative feedback, or even by trying out different permutations of pairs.
|
||||
This way you can give refreshing recommendations, while still being in control by providing positive and negative feedback, or even by trying out different permutations of pairs.
|
||||
|
||||
## Wrapping up
|
||||
|
||||
Discovery search is a powerful tool that lets you explore the vector space in a more controlled way. It can be used to find points that are not necessarily close to the target, but are still relevant to the search. It can also be used to represent complex tastes, and break out of the similarity bubble. Check out the [documentation](/documentation/concepts/explore/#discovery-api) to learn more about the math behind it and how to use it.
|
||||
## Key rakeaways:
|
||||
- Discovery search is a powerful tool for controlled exploration in vector spaces.
|
||||
Context, positive, and negative vectors guide search parameters and refine results.
|
||||
- Real-world applications include multimodal search, diverse recommendations, and context-driven exploration.
|
||||
- Ready to experience the power of Qdrant's Discovery search for yourself? [Try a free demo](https://qdrant.tech/contact-us/) now and unlock the full potential of controlled exploration in vector spaces!
|
||||
@@ -54,7 +54,7 @@ As embeddings are vectors, one can apply a simple function to calculate the simi
|
||||
So with similarity learning, all we need to do is provide pairs of correct questions and answers.
|
||||
And then, the model will learn to distinguish proper answers by the similarity of embeddings.
|
||||
|
||||
>If you want to learn more about similarity learning and applications, check out this [article](https://qdrant.tech/documentation/tutorials/neural-search/) which might be an asset.
|
||||
>If you want to learn more about similarity learning and applications, check out this [article](/documentation/tutorials/neural-search/) which might be an asset.
|
||||
|
||||
## Let's build
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "FastEmbed: Fast and Lightweight Embedding Generation for Text"
|
||||
title: "FastEmbed: Qdrant's Efficient Python Library for Embedding Generation"
|
||||
short_description: "FastEmbed: Quantized Embedding models for fast CPU Generation"
|
||||
description: "FastEmbed is a Python library engineered for speed, efficiency, and accuracy"
|
||||
description: "Learn how to accurately and efficiently create text embeddings with FastEmbed."
|
||||
social_preview_image: /articles_data/fastembed/preview/social_preview.jpg
|
||||
small_preview_image: /articles_data/fastembed/preview/lightning.svg
|
||||
preview_dir: /articles_data/fastembed/preview
|
||||
@@ -21,15 +21,15 @@ keywords:
|
||||
- quantized embedding model
|
||||
---
|
||||
|
||||
Data Science and Machine Learning practitioners often find themselves navigating through a labyrinth of models, libraries, and frameworks. Which model to choose, what embedding size, how to approach tokenizing, these are just some questions you are faced with when starting your work. We understood how, for many data scientists, they wanted an easier and intuitive means to do their embedding work. This is why we built FastEmbed (docs: https://qdrant.github.io/fastembed/) —a Python library engineered for speed, efficiency, and above all, usability. We have created easy to use default workflows, handling the 80% use cases in NLP embedding.
|
||||
Data Science and Machine Learning practitioners often find themselves navigating through a labyrinth of models, libraries, and frameworks. Which model to choose, what embedding size, and how to approach tokenizing, are just some questions you are faced with when starting your work. We understood how many data scientists wanted an easier and more intuitive means to do their embedding work. This is why we built FastEmbed, a Python library engineered for speed, efficiency, and usability. We have created easy to use default workflows, handling the 80% use cases in NLP embedding.
|
||||
|
||||
### Current State of Affairs for Generating Embeddings
|
||||
## Current State of Affairs for Generating Embeddings
|
||||
|
||||
Usually you make embedding by utilizing PyTorch or TensorFlow models under the hood. But using these libraries comes at a cost in terms of ease of use and computational speed. This is at least in part because these are built for both: model inference and improvement e.g. via fine-tuning.
|
||||
Usually you make embedding by utilizing PyTorch or TensorFlow models under the hood. However, using these libraries comes at a cost in terms of ease of use and computational speed. This is at least in part because these are built for both: model inference and improvement e.g. via fine-tuning.
|
||||
|
||||
To tackle these problems we built a small library focused on the task of quickly and efficiently creating text embeddings. We also decided to start with only a small sample of best in class transformer models. By keeping it small and focused on a particular use case, we could make our library focused without all the extraneous dependencies. We ship with limited models, quantize the model weights and seamlessly integrate them with the ONNX Runtime. FastEmbed strikes a balance between inference time, resource utilization and performance (recall/accuracy).
|
||||
|
||||
### Quick Example
|
||||
## Quick Embedding Text Document Example
|
||||
|
||||
Here is an example of how simple we have made embedding text documents:
|
||||
|
||||
@@ -75,7 +75,7 @@ Next, we initialize the Embedding model with the default model: [BAAI/bge-small
|
||||
embedding_model = DefaultEmbedding()
|
||||
```
|
||||
|
||||
The default model and several other models have a context window of maximum 512 tokens. This maximum limit comes from the embedding model training and design itself.If you'd like to embed sequences larger than that, we'd recommend using some pooling strategy to get a single vector out of the sequence. For example, you can use the mean of the embeddings of different chunks of a document. This is also what the [SBERT Paper recommends](https://lilianweng.github.io/posts/2021-05-31-contrastive/#sentence-bert)
|
||||
The default model and several other models have a context window of a maximum of 512 tokens. This maximum limit comes from the embedding model training and design itself. If you'd like to embed sequences larger than that, we'd recommend using some pooling strategy to get a single vector out of the sequence. For example, you can use the mean of the embeddings of different chunks of a document. This is also what the [SBERT Paper recommends](https://lilianweng.github.io/posts/2021-05-31-contrastive/#sentence-bert)
|
||||
|
||||
This model strikes a balance between speed and accuracy, ideal for real-world applications.
|
||||
|
||||
@@ -89,7 +89,7 @@ The `embed()` method returns a list of NumPy arrays, each corresponding to the
|
||||
|
||||
You can easily parse these NumPy arrays for any downstream application—be it clustering, similarity comparison, or feeding them into a machine learning model for further analysis.
|
||||
|
||||
## Key Features
|
||||
## 3 Key Features of FastEmbed
|
||||
|
||||
FastEmbed is built for inference speed, without sacrificing (too much) performance:
|
||||
|
||||
@@ -101,7 +101,7 @@ We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we've chosen that
|
||||
|
||||

|
||||
|
||||
## Under the Hood
|
||||
## Under the Hood of FastEmbed
|
||||
|
||||
**Quantized Models**: We quantize the models for CPU (and Mac Metal) – giving you the best buck for your compute model. Our default model is so small, you can run this in AWS Lambda if you’d like!
|
||||
|
||||
@@ -123,7 +123,7 @@ This minimized list serves two purposes. First, it significantly reduces the ins
|
||||
|
||||
Notably absent from the dependency list are bulky libraries like PyTorch, and there’s no requirement for CUDA drivers. This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups.
|
||||
|
||||
**ONNXRuntime**: The ONNXRuntime gives us the ability to support multiple providers. The quantization we do is limited for CPU (Intel), but we intend to support GPU versions of the same in future as well. This allows for greater customization and optimization, further aligning with your specific performance and computational requirements.
|
||||
**ONNXRuntime**: The ONNXRuntime gives us the ability to support multiple providers. The quantization we do is limited for CPU (Intel), but we intend to support GPU versions of the same in the future as well. This allows for greater customization and optimization, further aligning with your specific performance and computational requirements.
|
||||
|
||||
## Current Models
|
||||
|
||||
@@ -137,15 +137,15 @@ When it comes to FastEmbed's DefaultEmbedding model, we're committed to supporti
|
||||
|
||||
If anything changes, you'll see a new version number pop up, like going from 0.0.6 to 0.1. So, it's a good idea to lock in the FastEmbed version you're using to avoid surprises.
|
||||
|
||||
## Usage with Qdrant
|
||||
## Using FastEmbed with Qdrant
|
||||
|
||||
Qdrant is a Vector Store, offering a comprehensive, efficient, and scalable solution for modern machine learning and AI applications. Whether you are dealing with billions of data points, require a low latency performant vector solution, or specialized quantization methods – [Qdrant is engineered](/documentation/overview/) to meet those demands head-on.
|
||||
Qdrant is a Vector Store, offering comprehensive, efficient, and scalable [enterprise solutions](https://qdrant.tech/enterprise-solutions/) for modern machine learning and AI applications. Whether you are dealing with billions of data points, require a low latency performant [vector database solution](https://qdrant.tech/qdrant-vector-database/), or specialized quantization methods – [Qdrant is engineered](/documentation/overview/) to meet those demands head-on.
|
||||
|
||||
The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the DefaultEmbedding and use that with Qdrant.
|
||||
|
||||
Below is a detailed guide on how to get started with FastEmbed in conjunction with Qdrant.
|
||||
|
||||
### Installation
|
||||
### Step 1: Installation
|
||||
|
||||
Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip:
|
||||
|
||||
@@ -159,7 +159,7 @@ For those using zsh as their shell, you might encounter syntax issues. In such c
|
||||
pip install 'qdrant-client[fastembed]'
|
||||
```
|
||||
|
||||
### Initializing the Qdrant Client
|
||||
### Step 2: Initializing the Qdrant Client
|
||||
|
||||
After successful installation, the next step involves initializing the Qdrant Client. This can be done either in-memory or by specifying a database path:
|
||||
|
||||
@@ -169,7 +169,7 @@ from qdrant_client import QdrantClient
|
||||
client = QdrantClient(":memory:") # or QdrantClient(path="path/to/db")
|
||||
```
|
||||
|
||||
### Preparing Documents, Metadata, and IDs
|
||||
### Step 3: Preparing Documents, Metadata, and IDs
|
||||
|
||||
Once the client is initialized, prepare the text documents you wish to embed, along with any associated metadata and unique IDs:
|
||||
|
||||
@@ -194,7 +194,7 @@ docs = [
|
||||
]
|
||||
```
|
||||
|
||||
### Adding Documents to a Collection
|
||||
### Step 4: Adding Documents to a Collection
|
||||
|
||||
With your documents, metadata, and IDs ready, you can proceed to add these to a specified collection within Qdrant using the add method:
|
||||
|
||||
@@ -207,11 +207,11 @@ client.add(
|
||||
)
|
||||
```
|
||||
|
||||
Inside this function, Qdrant Client uses FastEmbed to make the text embedding, generate ids if they’re missing and then adding them to the index with metadata. This uses the DefaultEmbedding model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)
|
||||
Inside this function, Qdrant Client uses FastEmbed to make the text embedding, generate ids if they’re missing, and then add them to the index with metadata. This uses the DefaultEmbedding model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)
|
||||
|
||||

|
||||
|
||||
### Performing Queries
|
||||
### Step 5: Performing Queries
|
||||
|
||||
Finally, you can perform queries on your stored documents. Qdrant offers a robust querying capability, and the query results can be easily retrieved as follows:
|
||||
|
||||
@@ -229,7 +229,7 @@ Behind the scenes, we first convert the query_text to the embedding and use tha
|
||||
|
||||
By following these steps, you effectively utilize the combined capabilities of FastEmbed and Qdrant, thereby streamlining your embedding generation and retrieval tasks.
|
||||
|
||||
Qdrant is designed to handle large-scale datasets with billions of data points. Its architecture employs techniques like binary and scalar quantization for efficient storage and retrieval. When you inject FastEmbed’s CPU-first design and lightweight nature into this equation, you end up with a system that can scale seamlessly while maintaining low latency.
|
||||
Qdrant is designed to handle large-scale datasets with billions of data points. Its architecture employs techniques like [binary quantization](https://qdrant.tech/articles/binary-quantization/) and [scalar quantization](https://qdrant.tech/articles/scalar-quantization/) for efficient storage and retrieval. When you inject FastEmbed’s CPU-first design and lightweight nature into this equation, you end up with a system that can scale seamlessly while maintaining low latency.
|
||||
|
||||
## Summary
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "Question Answering with LangChain and Qdrant without boilerplate"
|
||||
title: "Using LangChain for Question Answering with Qdrant"
|
||||
short_description: "Large Language Models might be developed fast with modern tool. Here is how!"
|
||||
description: "We combined LangChain, pretrained LLM from OpenAI, SentenceTransformers and Qdrant to create a Q&A system with just a few lines of code."
|
||||
description: "We combined LangChain, a pre-trained LLM from OpenAI, SentenceTransformers & Qdrant to create a question answering system with just a few lines of code. Learn more!"
|
||||
social_preview_image: /articles_data/langchain-integration/social_preview.png
|
||||
small_preview_image: /articles_data/langchain-integration/chain.svg
|
||||
preview_dir: /articles_data/langchain-integration/preview
|
||||
@@ -20,20 +20,22 @@ keywords:
|
||||
- embeddings
|
||||
---
|
||||
|
||||
Building applications with Large Language Models don't have to be complicated. A lot has been going on recently to simplify the development,
|
||||
# Streamlining Question Answering: Simplifying Integration with LangChain and Qdrant
|
||||
|
||||
Building applications with Large Language Models doesn't have to be complicated. A lot has been going on recently to simplify the development,
|
||||
so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://langchain.readthedocs.io)
|
||||
provides unified interfaces to different libraries, so you can avoid writing boilerplate code and focus on the value you want to bring.
|
||||
|
||||
## Question Answering with Qdrant in the loop
|
||||
## Why Use Qdrant for Question Answering with LangChain?
|
||||
|
||||
It has been reported millions of times recently, but let's say that again. ChatGPT-like models struggle with generating factual statements if no context
|
||||
is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus, it is better to provide some facts we
|
||||
know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give a comprehensive answer. Vector database,
|
||||
such as Qdrant, is of great help here, as their ability to perform a semantic search over a huge knowledge base is crucial to preselect some possibly valid
|
||||
documents, so they can be provided into the LLM. That's also one of the **chains** implemented in LangChain, which is called `VectorDBQA`. And Qdrant got
|
||||
know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give a comprehensive answer. [Vector database,
|
||||
such as Qdrant](https://qdrant.tech/), is of great help here, as their ability to perform a [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid
|
||||
documents, so they can be provided into the LLM. That's also one of the **chains** implemented in [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which is called `VectorDBQA`. And Qdrant got
|
||||
integrated with the library, so it might be used to build it effortlessly.
|
||||
|
||||
### What do we need?
|
||||
### The Two-Model Approach
|
||||
|
||||
Surprisingly enough, there will be two models required to set things up. First of all, we need an embedding model that will convert the set of facts into
|
||||
vectors, and store those into Qdrant. That's an identical process to any other semantic search application. We're going to use one of the
|
||||
@@ -41,7 +43,7 @@ vectors, and store those into Qdrant. That's an identical process to any other s
|
||||
similar documents, given the query.
|
||||
|
||||
However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all
|
||||
of them into a single text. Then, we build a prompt to the LLM (in our case OpenAI), including those documents as a context, of course together with the
|
||||
of them into a single text. Then, we build a prompt to the LLM (in our case [OpenAI](https://openai.com/)), including those documents as a context, of course together with the
|
||||
question asked. So the input to the LLM looks like the following:
|
||||
|
||||
```text
|
||||
@@ -56,27 +58,28 @@ Helpful Answer:
|
||||
There might be several context documents combined, and it is solely up to LLM to choose the right piece of content. But our expectation is, the model should
|
||||
respond with just `4`.
|
||||
|
||||
Why do we need two different models? Both solve some different tasks. The first model performs feature extraction, by converting the text into vectors, while
|
||||
## Why do we need two different models?
|
||||
Both solve some different tasks. The first model performs feature extraction, by converting the text into vectors, while
|
||||
the second one helps in text generation or summarization. Disclaimer: This is not the only way to solve that task with LangChain. Such a chain is called `stuff`
|
||||
in the library nomenclature.
|
||||
|
||||

|
||||
|
||||
Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few lines
|
||||
of code, thanks to the recent integration with Qdrant. We're not even going to work directly with `QdrantClient`, as everything is already done in the background
|
||||
of code, thanks to the recent integration with [Qdrant](https://qdrant.tech/). We're not even going to work directly with `QdrantClient`, as everything is already done in the background
|
||||
by LangChain. If you want to get into the source code right away, all the processing is available as a
|
||||
[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing).
|
||||
|
||||
## Implementing Question Answering with LangChain and Qdrant
|
||||
## How to Implement Question Answering with LangChain and Qdrant
|
||||
|
||||
### Configuration
|
||||
### Step 1: Configuration
|
||||
|
||||
A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant Cloud](https://qdrant.tech),
|
||||
A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant Cloud](https://cloud.qdrant.io),
|
||||
so we need an API key. The same is for OpenAI - the API key has to be obtained from their website.
|
||||
|
||||

|
||||
|
||||
### Building the knowledge base
|
||||
### Step 2: Building the knowledge base
|
||||
|
||||
We also need some facts from which the answers will be generated. There is plenty of public datasets available, and
|
||||
[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them. It consists of the whole HTML content of the websites they were
|
||||
@@ -88,14 +91,14 @@ other options available. LangChain will handle that part of the process in a sin
|
||||
|
||||

|
||||
|
||||
### Setting up QA with Qdrant in a loop
|
||||
### Step 3: Setting up QA with Qdrant in a loop
|
||||
|
||||
`VectorDBQA` is a chain that performs the process described above. So it, first of all, loads some facts from Qdrant and then feeds them into OpenAI LLM which
|
||||
should analyze them to find the answer to a given question. The only last thing to do before using it is to put things together, also with a single function call.
|
||||
|
||||

|
||||
|
||||
## Testing out the chain
|
||||
## Step 4: Testing out the chain
|
||||
|
||||
And that's it! We can put some queries, and LangChain will perform all the required processing to find the answer in the provided context.
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: Minimal RAM you need to serve a million vectors
|
||||
short_description: How to properly measure RAM usage and optimize Qdrant for memory consumption.
|
||||
description: How to properly measure RAM usage and optimize Qdrant for memory consumption.
|
||||
title: "How to Optimize RAM Requirements for 1 Million Vectors: A Case Study"
|
||||
short_description: Master RAM measurement and memory optimization for optimal performance and resource use.
|
||||
description: Unlock the secrets of efficient RAM measurement and memory optimization with this comprehensive guide, ensuring peak performance and resource utilization.
|
||||
social_preview_image: /articles_data/memory-consumption/preview/social_preview.jpg
|
||||
preview_dir: /articles_data/memory-consumption/preview
|
||||
small_preview_image: /articles_data/memory-consumption/icon.svg
|
||||
@@ -32,6 +32,7 @@ Introduction:
|
||||
3. As a result, if you see `10Gb` memory consumption in `htop`, it doesn't mean that your process actually needs `10Gb` of RAM to work.
|
||||
-->
|
||||
|
||||
# Mastering RAM Measurement and Memory Optimization in Qdrant: A Comprehensive Guide
|
||||
|
||||
When it comes to measuring the memory consumption of our processes, we often rely on tools such as `htop` to give us an indication of how much RAM is being used. However, this method can be misleading and doesn't always accurately reflect the true memory usage of a process.
|
||||
|
||||
@@ -42,9 +43,9 @@ This means that the memory consumption of the child process will be counted twic
|
||||
Additionally, a process may utilize disk cache, which is also accounted as resident memory in the `htop` measurements.
|
||||
|
||||
As a result, even if `htop` shows that a process is using 10GB of memory, it doesn't necessarily mean that the process actually requires 10GB of RAM to operate efficiently.
|
||||
In this article, we will explore how to properly measure RAM usage and optimize Qdrant for optimal memory consumption.
|
||||
In this article, we will explore how to properly measure RAM usage and optimize [Qdrant](https://qdrant.tech/) for optimal memory consumption.
|
||||
|
||||
## How to measure actual memory requirements
|
||||
## How to measure actual RAM requirements
|
||||
|
||||
<!--
|
||||
1. We need to know how much RAM we need to have for the program to work, so why not just do a straightforward experiment.
|
||||
@@ -62,7 +63,7 @@ In this article, we will explore how to properly measure RAM usage and optimize
|
||||
|
||||
-->
|
||||
|
||||
We need to know memory consumption in order to estimate how much RAM we need to run the program.
|
||||
We need to know memory consumption in order to estimate how much RAM is required to run the program.
|
||||
So in order to determine that, we can conduct a simple experiment.
|
||||
Let's limit the allowed memory of the process and observe at which point it stops functioning.
|
||||
In this way we can determine the minimum amount of RAM the program needs to operate.
|
||||
@@ -175,9 +176,9 @@ Create collection with:
|
||||
```http
|
||||
PUT /collections/benchmark
|
||||
{
|
||||
...
|
||||
"optimizers_config": {
|
||||
"mmap_threshold_kb": 20000
|
||||
"vectors": {
|
||||
...
|
||||
"on_disk": true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -207,8 +208,6 @@ Now the out-of-memory happens when we allow using **600mb** RAM only
|
||||
|
||||
</details>
|
||||
|
||||
<br/>
|
||||
|
||||
At this point we have to switch from network-mounted storage to a faster disk, as the network-based storage is too slow to handle the amount of sequential reads that our system needs to serve the queries.
|
||||
|
||||
But let's first see how much RAM we need to serve 1 million vectors and then we will discuss the speed optimization as well.
|
||||
@@ -216,19 +215,20 @@ But let's first see how much RAM we need to serve 1 million vectors and then we
|
||||
|
||||
### Vectors and HNSW graph stored using MMAP
|
||||
|
||||
In the third experiment, we tested how well our system performs when vectors and HNSW graph are stored using the memory-mapped files.
|
||||
In the third experiment, we tested how well our system performs when vectors and [HNSW](https://qdrant.tech/articles/filtrable-hnsw/) graph are stored using the memory-mapped files.
|
||||
Create collection with:
|
||||
|
||||
```http
|
||||
PUT /collections/benchmark
|
||||
{
|
||||
...
|
||||
"vectors": {
|
||||
...
|
||||
"on_disk": true
|
||||
},
|
||||
"hnsw_config": {
|
||||
"on_disk": true
|
||||
},
|
||||
"optimizers_config": {
|
||||
"mmap_threshold_kb": 20000
|
||||
}
|
||||
...
|
||||
}
|
||||
```
|
||||
|
||||
@@ -249,8 +249,6 @@ With this configuration we are able to serve 1 million vectors with **only 135mb
|
||||
|
||||
</details>
|
||||
|
||||
<br/>
|
||||
|
||||
At this point the importance of the disk speed becomes critical.
|
||||
We can serve the search requests with 135mb of RAM, but the speed of the requests makes it impossible to use the system in production.
|
||||
|
||||
@@ -357,8 +355,7 @@ Which might be an interesting option to serve large datasets with low search lat
|
||||
|
||||
## Conclusion
|
||||
|
||||
In this article, we showed that Qdrant have flexibility in terms of RAM usage and can be used to serve large datasets.
|
||||
It provides configurable trade-offs between RAM usage and search speed.
|
||||
In this article, we showed that Qdrant has flexibility in terms of RAM usage and can be used to serve large datasets. It provides configurable trade-offs between RAM usage and search speed. If you’re interested to learn more about Qdrant, [book a demo today](https://qdrant.tech/contact-us/)!
|
||||
|
||||
We are eager to learn more about how you use Qdrant in your projects, what challenges you face, and how we can help you solve them.
|
||||
Please feel free to join our [Discord](https://qdrant.to/discord) and share your experience with us!
|
||||
|
||||
@@ -216,7 +216,7 @@ Qdrant has a pre-built docker image and start working with it is just as simple
|
||||
docker run -p 6333:6333 qdrant/qdrant
|
||||
```
|
||||
|
||||
Documentation with examples could be found [here](https://qdrant.github.io/qdrant/redoc/index.html).
|
||||
Documentation with examples could be found [here](https://api.qdrant.tech/api-reference).
|
||||
|
||||
|
||||
## Conclusion
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "Best Practices for Massive-Scale Deployments: Multitenancy and Custom Sharding"
|
||||
short_description: "Combining our most popular features to support scalable machine learning solutions."
|
||||
description: "Combining our most popular features to support scalable machine learning solutions."
|
||||
title: "How to Implement Multitenancy and Custom Sharding in Qdrant"
|
||||
short_description: "Explore how Qdrant's multitenancy and custom sharding streamline machine-learning operations, enhancing scalability and data security."
|
||||
description: "Discover how multitenancy and custom sharding in Qdrant can streamline your machine-learning operations. Learn how to scale efficiently and manage data securely."
|
||||
social_preview_image: /articles_data/multitenancy/social_preview.png
|
||||
preview_dir: /articles_data/multitenancy/preview
|
||||
small_preview_image: /articles_data/multitenancy/icon.svg
|
||||
@@ -16,11 +16,15 @@ keywords:
|
||||
- vector database
|
||||
---
|
||||
|
||||
# Scaling Your Machine Learning Setup: The Power of Multitenancy and Custom Sharding in Qdrant
|
||||
|
||||
We are seeing the topics of [multitenancy](/documentation/guides/multiple-partitions/) and [distributed deployment](/documentation/guides/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup.
|
||||
|
||||
Whether you are building a bank fraud-detection system, RAG for e-commerce, or services for the federal government - you will need to leverage a multitenant architecture to scale your product.
|
||||
Whether you are building a bank fraud-detection system, [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) for e-commerce, or services for the federal government - you will need to leverage a multitenant architecture to scale your product.
|
||||
In the world of SaaS and enterprise apps, this setup is the norm. It will considerably increase your application's performance and lower your hosting costs.
|
||||
|
||||
## Multitenancy & custom sharding with Qdrant
|
||||
|
||||
We have developed two major features just for this. __You can now scale a single Qdrant cluster and support all of your customers worldwide.__ Under [multitenancy](/documentation/guides/multiple-partitions/), each customer's data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer's access. This is called [custom sharding](/documentation/guides/distributed_deployment/#user-defined-sharding).
|
||||
|
||||
Combining these two will result in an efficiently-partitioned architecture that further leverages the convenience of a single Qdrant cluster. This article will briefly explain the benefits and show how you can get started using both features.
|
||||
@@ -175,9 +179,9 @@ client.create_payload_index(
|
||||
```
|
||||
> Note: Keep in mind that global requests (without the `group_id` filter) will be slower since they will necessitate scanning all groups to identify the nearest neighbors.
|
||||
|
||||
## Next steps
|
||||
## Explore multitenancy and custom sharding in Qdrant for scalable solutions
|
||||
|
||||
Qdrant is ready to support a massive-scale architecture for your machine learning project. If you want to see whether our vector database is right for you, try the [quickstart tutorial](/documentation/quick-start/) or read our [docs and tutorials](/documentation/).
|
||||
Qdrant is ready to support a massive-scale architecture for your machine learning project. If you want to see whether our [vector database](https://qdrant.tech/) is right for you, try the [quickstart tutorial](/documentation/quick-start/) or read our [docs and tutorials](/documentation/).
|
||||
|
||||
To spin up a free instance of Qdrant, sign up for [Qdrant Cloud](https://qdrant.to/cloud) - no strings attached.
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: Neural Search Tutorial
|
||||
title: "Neural Search 101: A Complete Guide and Step-by-Step Tutorial"
|
||||
short_description: Step-by-step guide on how to build a neural search service.
|
||||
description: Our step-by-step guide on how to build a neural search service with BERT + Qdrant + FastAPI.
|
||||
description: Discover the power of neural search. Learn what neural search is and follow our tutorial to build a neural search service using BERT, Qdrant, and FastAPI.
|
||||
# external_link: https://blog.qdrant.tech/neural-search-tutorial-3f034ab13adc
|
||||
social_preview_image: /articles_data/neural-search-tutorial/social_preview.jpg
|
||||
preview_dir: /articles_data/neural-search-tutorial/preview
|
||||
@@ -12,15 +12,14 @@ author_link: https://blog.vasnetsov.com/
|
||||
date: 2021-06-10T10:18:00.000Z
|
||||
# aliases: [ /articles/neural-search-tutorial/ ]
|
||||
---
|
||||
|
||||
## How to build a neural search service with BERT + Qdrant + FastAPI
|
||||
# Neural Search 101: A Comprehensive Guide and Step-by-Step Tutorial
|
||||
|
||||
Information retrieval technology is one of the main technologies that enabled the modern Internet to exist.
|
||||
These days, search technology is the heart of a variety of applications.
|
||||
From web-pages search to product recommendations.
|
||||
For many years, this technology didn't get much change until neural networks came into play.
|
||||
|
||||
In this tutorial we are going to find answers to these questions:
|
||||
In this guide we are going to find answers to these questions:
|
||||
|
||||
* What is the difference between regular and neural search?
|
||||
* What neural networks could be used for search?
|
||||
@@ -49,7 +48,7 @@ The usual Euclidean distance can also be used, but it is not so efficient due to
|
||||
|
||||
It is ideal to use a model specially trained to determine the closeness of meanings.
|
||||
For example, models trained on Semantic Textual Similarity (STS) datasets.
|
||||
Current state-of-the-art models could be found on this [leaderboard](https://paperswithcode.com/sota/semantic-textual-similarity-on-sts-benchmark?p=roberta-a-robustly-optimized-bert-pretraining).
|
||||
Current state-of-the-art models can be found on this [leaderboard](https://paperswithcode.com/sota/semantic-textual-similarity-on-sts-benchmark?p=roberta-a-robustly-optimized-bert-pretraining).
|
||||
|
||||
However, not only specially trained models can be used.
|
||||
If the model is trained on a large enough dataset, its internal features can work as embeddings too.
|
||||
@@ -60,7 +59,7 @@ The output of this layer can be used as an embedding.
|
||||
## What tasks is neural search good for?
|
||||
|
||||
Neural search has the greatest advantage in areas where the query cannot be formulated precisely.
|
||||
Querying a table in a SQL database is not the best place for neural search.
|
||||
Querying a table in an SQL database is not the best place for neural search.
|
||||
|
||||
On the contrary, if the query itself is fuzzy, or it cannot be formulated as a set of conditions - neural search can help you.
|
||||
If the search query is a picture, sound file or long text, neural network search is almost the only option.
|
||||
@@ -69,7 +68,7 @@ If you want to build a recommendation system, the neural approach can also be us
|
||||
The user's actions can be encoded in vector space in the same way as a picture or text.
|
||||
And having those vectors, it is possible to find semantically similar users and determine the next probable user actions.
|
||||
|
||||
## Let's build our own
|
||||
## Step-by-step neural search tutorial using Qdrant
|
||||
|
||||
With all that said, let's make our neural network search.
|
||||
As an example, I decided to make a search for startups by their description.
|
||||
@@ -80,7 +79,7 @@ I will use data from [startups-list.com](https://www.startups-list.com/).
|
||||
Each record contains the name, a paragraph describing the company, the location and a picture.
|
||||
Raw parsed data can be found at [this link](https://storage.googleapis.com/generall-shared-data/startups_demo.json).
|
||||
|
||||
### Prepare data for neural search
|
||||
### Step 1: Prepare data for neural search
|
||||
|
||||
To be able to search for our descriptions in vector space, we must get vectors first.
|
||||
We need to encode the descriptions into a vector representation.
|
||||
@@ -99,7 +98,7 @@ The complete code for data preparation with detailed comments can be found and r
|
||||
|
||||
[](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing)
|
||||
|
||||
### Vector search engine
|
||||
### Step 2: Incorporate a Vector search engine
|
||||
|
||||
Now as we have a vector representation for all our records, we need to store them somewhere.
|
||||
In addition to storing, we may also need to add or delete a vector, save additional information with the vector.
|
||||
@@ -107,9 +106,9 @@ And most importantly, we need a way to search for the nearest vectors.
|
||||
|
||||
The vector search engine can take care of all these tasks.
|
||||
It provides a convenient API for searching and managing vectors.
|
||||
In our tutorial we will use [Qdrant](https://github.com/qdrant/qdrant) vector search engine.
|
||||
It not only supports all necessary operations with vectors but also allows to store additional payload along with vectors and use it to perform filtering of the search result.
|
||||
Qdrant has a client for python and also defines the API schema if you need to use it from other languages.
|
||||
In our tutorial, we will use [Qdrant vector search engine](https://github.com/qdrant/qdrant) vector search engine.
|
||||
It not only supports all necessary operations with vectors but also allows you to store additional payload along with vectors and use it to perform filtering of the search result.
|
||||
Qdrant has a client for Python and also defines the API schema if you need to use it from other languages.
|
||||
|
||||
The easiest way to use Qdrant is to run a pre-built image.
|
||||
So make sure you have Docker installed on your system.
|
||||
@@ -142,7 +141,7 @@ To make sure you can test [http://localhost:6333/](http://localhost:6333/) in yo
|
||||
|
||||
All uploaded to Qdrant data is saved into the `./qdrant_storage` directory and will be persisted even if you recreate the container.
|
||||
|
||||
### Upload data to Qdrant
|
||||
### Step 3: Upload data to Qdrant
|
||||
|
||||
Now once we have the vectors prepared and the search engine running, we can start uploading the data.
|
||||
To interact with Qdrant from python, I recommend using an out-of-the-box client library.
|
||||
@@ -219,12 +218,12 @@ qdrant_client.upload_collection(
|
||||
)
|
||||
```
|
||||
|
||||
Now we have vectors, uploaded to the vector search engine.
|
||||
On the next step we will learn how to actually search for closest vectors.
|
||||
Now we have vectors uploaded to the vector search engine.
|
||||
In the next step, we will learn how to actually search for the closest vectors.
|
||||
|
||||
The full code for this step could be found [here](https://github.com/qdrant/qdrant_demo/blob/master/qdrant_demo/init_collection_startups.py).
|
||||
The full code for this step can be found [here](https://github.com/qdrant/qdrant_demo/blob/master/qdrant_demo/init_collection_startups.py).
|
||||
|
||||
### Make a search API
|
||||
### Step 4: Make a search API
|
||||
|
||||
Now that all the preparations are complete, let's start building a neural search class.
|
||||
|
||||
@@ -306,7 +305,7 @@ from qdrant_client.models import Filter
|
||||
We now have a class for making neural search queries. Let's wrap it up into a service.
|
||||
|
||||
|
||||
### Deploy as a service
|
||||
### Step 5: Deploy as a service
|
||||
|
||||
To build the service we will use the FastAPI framework.
|
||||
It is super easy to use and requires minimal code writing.
|
||||
@@ -359,17 +358,10 @@ Feel free to play around with it, make queries and check out the results.
|
||||
This concludes the tutorial.
|
||||
|
||||
|
||||
### Online Demo
|
||||
### Experience Neural Search With Qdrant’s Free Demo
|
||||
Excited to see neural search in action? Take the next step and book a [free demo](https://qdrant.to/semantic-search-demo) with Qdrant! Experience firsthand how this cutting-edge technology can transform your search capabilities.
|
||||
|
||||
The described code is the core of this [online demo](https://qdrant.to/semantic-search-demo).
|
||||
You can try it to get an intuition for cases when the neural search is useful.
|
||||
The demo contains a switch that selects between neural and full-text searches.
|
||||
You can turn neural search on and off to compare the result with regular full-text search.
|
||||
Try to use startup description to find similar ones.
|
||||
Our demo will help you grow intuition for cases when the neural search is useful. The demo contains a switch that selects between neural and full-text searches. You can turn neural search on and off to compare the result with regular full-text search.
|
||||
Try to use a startup description to find similar ones.
|
||||
|
||||
## Conclusion
|
||||
|
||||
In this tutorial, I have tried to give minimal information about neural search, but enough to start using it.
|
||||
Many potential applications are not mentioned here, this is a space to go further into the subject.
|
||||
|
||||
Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, publish other examples of neural networks and neural search applications.
|
||||
Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, and publish other examples of neural networks and neural search applications.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "Qdrant under the hood: Product Quantization"
|
||||
title: "Product Quantization in Vector Search | Qdrant"
|
||||
short_description: "Vector search with low memory? Try out our brand-new Product Quantization!"
|
||||
description: "Vector search with low memory? Try out our brand-new Product Quantization!"
|
||||
description: "Discover product quantization in vector search technology. Learn how it optimizes storage and accelerates search processes for high-dimensional data."
|
||||
social_preview_image: /articles_data/product-quantization/social_preview.png
|
||||
small_preview_image: /articles_data/product-quantization/product-quantization-icon.svg
|
||||
preview_dir: /articles_data/product-quantization/preview
|
||||
@@ -17,20 +17,23 @@ keywords:
|
||||
aliases: [ /articles/product_quantization/ ]
|
||||
---
|
||||
|
||||
# Product Quantization Demystified: Streamlining Efficiency in Data Management
|
||||
|
||||
Qdrant 1.1.0 brought the support of [Scalar Quantization](/articles/scalar-quantization/),
|
||||
a technique of reducing the memory footprint by even four times, by using `int8` to represent
|
||||
the values that would be normally represented by `float32`.
|
||||
|
||||
The memory usage in vector search might be reduced even further! Please welcome **Product
|
||||
The memory usage in [vector search](https://qdrant.tech/solutions/) might be reduced even further! Please welcome **Product
|
||||
Quantization**, a brand-new feature of Qdrant 1.2.0!
|
||||
|
||||
## Product Quantization
|
||||
## What is Product Quantization?
|
||||
|
||||
Product Quantization converts floating-point numbers into integers like every other quantization
|
||||
method. However, the process is slightly more complicated than Scalar Quantization and is more
|
||||
customizable, so you can find the sweet spot between memory usage and search precision. This article
|
||||
method. However, the process is slightly more complicated than [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/) and is more customizable, so you can find the sweet spot between memory usage and search precision. This article
|
||||
covers all the steps required to perform Product Quantization and the way it's implemented in Qdrant.
|
||||
|
||||
## How Does Product Quantization Work?
|
||||
|
||||
Let’s assume we have a few vectors being added to the collection and that our optimizer decided
|
||||
to start creating a new segment.
|
||||
|
||||
@@ -94,7 +97,7 @@ distance between a query and all the centroids.
|
||||
| **Chunk 1** | 0.08421 | 0.00142 | |
|
||||
| **...** | ... | ... | ... |
|
||||
|
||||
## Benchmarks
|
||||
## Produc Quantization Benchmarks
|
||||
|
||||
Product Quantization comes with a cost - there are some additional operations to perform so
|
||||
that the performance might be reduced. However, memory usage might be reduced drastically as
|
||||
@@ -205,9 +208,9 @@ the lower the search precision. The main benefit is undoubtedly the reduced usag
|
||||
It turns out that in some cases, Product Quantization may not only reduce the memory usage,
|
||||
but also the search time.
|
||||
|
||||
## Good practices
|
||||
## Product Quantization vs Scalar Quantization
|
||||
|
||||
Compared to Scalar Quantization, Product Quantization offers a higher compression rate. However, this comes with considerable trade-offs in accuracy, and at times, in-RAM search speed.
|
||||
Compared to [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/), Product Quantization offers a higher compression rate. However, this comes with considerable trade-offs in accuracy, and at times, in-RAM search speed.
|
||||
|
||||
Product Quantization tends to be favored in certain specific scenarios:
|
||||
|
||||
@@ -217,6 +220,10 @@ Product Quantization tends to be favored in certain specific scenarios:
|
||||
|
||||
In circumstances that do not align with the above, Scalar Quantization should be the preferred choice.
|
||||
|
||||
Qdrant documentation on [Product Quantization](/documentation/guides/quantization/#setting-up-product-quantization)
|
||||
will help you to set and configure the new quantization for your data and achieve even
|
||||
## Using Qdrant for Product Quantization
|
||||
|
||||
|
||||
If you’re already a Qdrant user, we have, documentation on [Product Quantization](/documentation/guides/quantization/#setting-up-product-quantization) that will help you to set and configure the new quantization for your data and achieve even
|
||||
up to 64x memory reduction.
|
||||
|
||||
Ready to experience the power of Product Quantization? [Sign up now](https://cloud.qdrant.io/) for a free Qdrant demo and optimize your data management today!
|
||||
@@ -27,7 +27,7 @@ set up your collections.
|
||||
|
||||
Previously, you had to send multiple requests to the Qdrant API to perform multiple non-related tasks. However, this
|
||||
can cause significant network overhead and slow down the process, especially if you have a poor connection speed.
|
||||
Fortunately, the [new batch search feature](https://qdrant.tech/documentation/concepts/search/#batch-search-api) allows
|
||||
Fortunately, the [new batch search feature](/documentation/concepts/search/#batch-search-api) allows
|
||||
you to avoid this issue. With just one API call, Qdrant will handle multiple search requests in the most efficient way
|
||||
possible. This means that you can perform multiple tasks simultaneously without having to worry about network overhead
|
||||
or slow performance.
|
||||
@@ -44,6 +44,6 @@ both ARM and non-ARM architectures using similar setups to understand the potent
|
||||
|
||||
Qdrant is a vector database that allows you to quickly search for the nearest neighbors. However, you may need to apply
|
||||
additional filters on top of the semantic search. Up until version 0.10, Qdrant only supported keyword filters. With the
|
||||
release of Qdrant 0.10, [you can now use full-text filters](https://qdrant.tech/documentation/concepts/filtering/#full-text-match)
|
||||
release of Qdrant 0.10, [you can now use full-text filters](/documentation/concepts/filtering/#full-text-match)
|
||||
as well. This new filter type can be used on its own or in combination with other filter types to provide even more
|
||||
flexibility in your searches.
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
---
|
||||
title: "Qdrant 1.8.0 - Major Performance Enhancements"
|
||||
title: "Qdrant 1.8.0: Enhanced Search Capabilities for Better Results"
|
||||
draft: false
|
||||
slug: qdrant-1.8.x
|
||||
short_description: "Faster sparse vectors.Optimized indexation. Optional CPU resource management."
|
||||
description: "Much faster sparse vectors, optimized indexation of text fields and optional CPU resource management configuration. "
|
||||
description: "Explore the latest in search technology with Qdrant 1.8.0! Discover faster performance, smarter indexing, and enhanced search capabilities."
|
||||
social_preview_image: /articles_data/qdrant-1.8.x/social_preview.png
|
||||
small_preview_image: /articles_data/qdrant-1.8.x/icon.svg
|
||||
preview_dir: /articles_data/qdrant-1.8.x/preview
|
||||
@@ -20,12 +20,14 @@ tags:
|
||||
- text field index
|
||||
---
|
||||
|
||||
[Qdrant 1.8.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.8.0).
|
||||
This time around, we have focused on Qdrant's internals. Our goal was to optimize performance, so that your existing setup can run faster and save on compute. Here is what we've been up to:
|
||||
# Unlocking Next-Level Search: Exploring Qdrant 1.8.0's Advanced Search Capabilities
|
||||
|
||||
- **Faster sparse vectors:** Hybrid search is up to 16x faster now!
|
||||
[Qdrant 1.8.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.8.0).
|
||||
This time around, we have focused on Qdrant's internals. Our goal was to optimize performance so that your existing setup can run faster and save on compute. Here is what we've been up to:
|
||||
|
||||
- **Faster [sparse vectors](https://qdrant.tech/articles/sparse-vectors/):** [Hybrid search](https://qdrant.tech/articles/hybrid-search/) is up to 16x faster now!
|
||||
- **CPU resource management:** You can allocate CPU threads for faster indexing.
|
||||
- **Better indexing performance:** We optimized text indexing on the backend.
|
||||
- **Better indexing performance:** We optimized text [indexing](https://qdrant.tech/documentation/concepts/indexing/) on the backend.
|
||||
|
||||
## Faster search with sparse vectors
|
||||
|
||||
@@ -36,7 +38,7 @@ What this means for your setup:
|
||||
- **Query speed:** The time it takes to run a search query has been significantly reduced.
|
||||
- **Search capacity:** Qdrant can now handle a much larger volume of search requests.
|
||||
- **User experience:** Results will appear faster, leading to a smoother experience for the user.
|
||||
- **Scalability:** You can easily accomodate rapidly growing users or an expanding dataset.
|
||||
- **Scalability:** You can easily accommodate rapidly growing users or an expanding dataset.
|
||||
|
||||
### Sparse vectors benchmark
|
||||
|
||||
@@ -49,7 +51,7 @@ Latency (y-axis) has dropped significantly for queries. You can see the before/a
|
||||

|
||||
**Figure 1:** Dropping latency in sparse vector search queries across versions 1.7-1.8.
|
||||
|
||||
The colors within both scatter plots show the frequency of results. The red dots show that the highest concentration is around 2200ms (before) and 135ms (after). This tells us that latency for sparse vectors queries dropped by about a factor of 16. Therefore, the time it takes to retrieve an answer with Qdrant is that much shorter.
|
||||
The colors within both scatter plots show the frequency of results. The red dots show that the highest concentration is around 2200ms (before) and 135ms (after). This tells us that latency for sparse vector queries dropped by about a factor of 16. Therefore, the time it takes to retrieve an answer with Qdrant is that much shorter.
|
||||
|
||||
This performance increase can have a dramatic effect on hybrid search implementations. [Read more about how to set this up.](/articles/sparse-vectors/)
|
||||
|
||||
@@ -76,7 +78,7 @@ optimizer_cpu_budget: 0
|
||||
|
||||
For most users, the default `optimizer_cpu_budget` setting will work well. We only recommend you use this if your indexing load is significant.
|
||||
|
||||
Our backend leverages dynamic CPU saturation to increase indexing speed. For that reason, the impact on search query performance ends up being minimal. Ultimately, you will be able to strike a the best possible balance between indexing times and search performance.
|
||||
Our backend leverages dynamic CPU saturation to increase indexing speed. For that reason, the impact on search query performance ends up being minimal. Ultimately, you will be able to strike the best possible balance between indexing times and search performance.
|
||||
|
||||
This configuration can be done at any time, but it requires a restart of Qdrant. Changing it affects both existing and new collections.
|
||||
|
||||
@@ -84,13 +86,13 @@ This configuration can be done at any time, but it requires a restart of Qdrant.
|
||||
|
||||
## Better indexing for text data
|
||||
|
||||
In order to minimize your RAM expenditure, we have developed a new way to index specific types of data. Please keep in mind that this is a backend improvement, and you won't need to configure anything.
|
||||
In order to [minimize your RAM expenditure](https://qdrant.tech/articles/memory-consumption/), we have developed a new way to index specific types of data. Please keep in mind that this is a backend improvement, and you won't need to configure anything.
|
||||
|
||||
> Going forward, if you are indexing immutable text fields, we estimate a 10% reduction in RAM loads. Our benchmark result is based on a system that uses 64GB of RAM. If you are using less RAM, this reduction might be higher than 10%.
|
||||
|
||||
Immutable text fields are static and do not change once they are added to Qdrant. These entries usually represent some type of an attribute, description or a tag. Vectors associated with them can be indexed more efficiently, since you don’t need to re-index them anymore. Conversely, mutable fields are dynamic and can be modified after their initial creation. Please keep in mind that they will continue to require additional RAM.
|
||||
Immutable text fields are static and do not change once they are added to Qdrant. These entries usually represent some type of attribute, description or tag. Vectors associated with them can be indexed more efficiently, since you don’t need to re-index them anymore. Conversely, mutable fields are dynamic and can be modified after their initial creation. Please keep in mind that they will continue to require additional RAM.
|
||||
|
||||
This approach ensures stability in the vector search index, with faster and more consistent operations. We achieved this by setting up a field index which helps minimize what is stored. To improve search performance we have also optimized the way we load documents for searches with a text field index. Now our backend loads documents mostly sequentially and in increasing order.
|
||||
This approach ensures stability in the [vector search](https://qdrant.tech/documentation/overview/vector-search/) index, with faster and more consistent operations. We achieved this by setting up a field index which helps minimize what is stored. To improve search performance we have also optimized the way we load documents for searches with a text field index. Now our backend loads documents mostly sequentially and in increasing order.
|
||||
|
||||
|
||||
## Minor improvements and new features
|
||||
@@ -103,7 +105,11 @@ Beyond these enhancements, [Qdrant v1.8.0](https://github.com/qdrant/qdrant/rele
|
||||
4. **Find points** whose payloads match more than the minimal amount of conditions. We included the `min_should` match feature for a condition to be `true` ([PR#3331](https://github.com/qdrant/qdrant/pull/3466/)).
|
||||
5. **Modify nested fields:** We have improved the `set_payload` API, adding the ability to update nested fields ([PR#3548](https://github.com/qdrant/qdrant/pull/3548)).
|
||||
|
||||
## Experience the Power of Qdrant 1.8.0
|
||||
|
||||
Ready to experience the enhanced performance of Qdrant 1.8.0? Upgrade now and explore the major improvements, from faster sparse vectors to optimized CPU resource management and better indexing for text data. Take your search capabilities to the next level with Qdrant's latest version. [Try a demo today](https://qdrant.tech/demo/) and see the difference firsthand!
|
||||
|
||||
## Release notes
|
||||
|
||||
For more information, see [our release notes](https://github.com/qdrant/qdrant/releases/tag/v1.8.0).
|
||||
Qdrant is an open source project. We welcome your contributions; raise [issues](https://github.com/qdrant/qdrant/issues), or contribute via [pull requests](https://github.com/qdrant/qdrant/pulls)!
|
||||
Qdrant is an open-source project. We welcome your contributions; raise [issues](https://github.com/qdrant/qdrant/issues), or contribute via [pull requests](https://github.com/qdrant/qdrant/pulls)!
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
{}
|
||||
@@ -1 +0,0 @@
|
||||
{"state":{"hard_state":{"term":0,"vote":0,"commit":0},"conf_state":{"voters":[3820286201499313],"learners":[],"voters_outgoing":[],"learners_next":[],"auto_leave":false}},"latest_snapshot_meta":{"term":0,"index":0},"apply_progress_queue":null,"peer_address_by_id":{},"this_peer_id":3820286201499313}
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: RAG is Dead. Long Live RAG!
|
||||
short_description: Why are vector databases needed for RAG? We debunk claims of increased LLM accuracy and look into drawbacks of large context windows.
|
||||
description: Why are vector databases needed for RAG? We debunk claims of increased LLM accuracy and look into drawbacks of large context windows.
|
||||
title: "Is RAG Dead? The Role of Vector Databases in Vector Search | Qdrant"
|
||||
short_description: Learn how Qdrant’s vector database enhances enterprise AI with superior accuracy and cost-effectiveness.
|
||||
description: Uncover the necessity of vector databases for RAG and learn how Qdrant's vector database empowers enterprise AI with unmatched accuracy and cost-effectiveness.
|
||||
social_preview_image: /articles_data/rag-is-dead/preview/social_preview.jpg
|
||||
small_preview_image: /articles_data/rag-is-dead/icon.svg
|
||||
preview_dir: /articles_data/rag-is-dead/preview
|
||||
@@ -17,11 +17,13 @@ keywords:
|
||||
- gemini 1.5
|
||||
---
|
||||
|
||||
When Anthropic came out with a context window of 100K tokens, they said: “*Vector search is dead. LLMs are getting more accurate and won’t need RAG anymore.*”
|
||||
# Is RAG Dead? The Role of Vector Databases in AI Efficiency and Vector Search
|
||||
|
||||
When Anthropic came out with a context window of 100K tokens, they said: “*[Vector search](https://qdrant.tech/solutions/) is dead. LLMs are getting more accurate and won’t need RAG anymore.*”
|
||||
|
||||
Google’s Gemini 1.5 now offers a context window of 10 million tokens. [Their supporting paper](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf) claims victory over accuracy issues, even when applying Greg Kamradt’s [NIAH methodology](https://twitter.com/GregKamradt/status/1722386725635580292).
|
||||
|
||||
*It’s over. RAG must be completely obsolete now. Right?*
|
||||
*It’s over. [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) (Retrieval Augmented Generation) must be completely obsolete now. Right?*
|
||||
|
||||
No.
|
||||
|
||||
@@ -29,25 +31,25 @@ Larger context windows are never the solution. Let me repeat. Never. They requir
|
||||
|
||||
The community is already stress testing Gemini 1.5:
|
||||
|
||||

|
||||

|
||||
|
||||
This is not surprising. LLMs require massive amounts of compute and memory to run. To cite Grant, running such a model by itself “would deplete a small coal mine to generate each completion”. Also, who is waiting 30 seconds for a response?
|
||||
|
||||
## Context stuffing is not the solution
|
||||
|
||||
> Relying on context is expensive, and it doesn’t improve response quality in real-world applications. Retrieval based on vector search offers much higher precision.
|
||||
> Relying on context is expensive, and it doesn’t improve response quality in real-world applications. Retrieval based on [vector search](https://qdrant.tech/solutions/) offers much higher precision.
|
||||
|
||||
If you solely rely on an LLM to perfect retrieval and precision, you are doing it wrong.
|
||||
If you solely rely on an [LLM](https://qdrant.tech/articles/what-is-rag-in-ai/) to perfect retrieval and precision, you are doing it wrong.
|
||||
|
||||
A large context window makes it harder to focus on relevant information. This increases the risk of errors or hallucinations in its responses.
|
||||
|
||||
Google found Gemini 1.5 significantly more accurate than GPT-4 at shorter context lengths and “a very small decrease in recall towards 1M tokens”. The recall is still below 0.8.
|
||||
|
||||

|
||||

|
||||
|
||||
We don’t think 60-80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information.
|
||||
|
||||
> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A vector database keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all.
|
||||
> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector database](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all.
|
||||
|
||||
Qdrant’s benchmark results are strongly in favor of accuracy and efficiency. We recommend that you consider them before deciding that an LLM is enough. Take a look at our [open-source benchmark reports](/benchmarks/) and [try out the tests](https://github.com/qdrant/vector-db-benchmark) yourself.
|
||||
|
||||
@@ -57,14 +59,14 @@ The future of AI lies in careful system engineering. As per [Zaharia et al.](htt
|
||||
|
||||
Even Gemini 1.5 demonstrates the need for a complex strategy. When looking at [Google’s MMLU Benchmark](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf), the model was called 32 times to reach a score of 90.0% accuracy. This shows us that even a basic compound arrangement is superior to monolithic models.
|
||||
|
||||
As a retrieval system, a vector database perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run.
|
||||
As a retrieval system, a [vector database](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run.
|
||||
|
||||
> The key advantage of RAG is that it allows an LLM to pull in real-time information from up-to-date internal and external knowledge sources, making it more dynamic and adaptable to new information. - Oliver Molander, CEO of IMAGINAI
|
||||
>
|
||||
|
||||
## Qdrant scales to enterprise RAG scenarios
|
||||
|
||||
People still don’t understand the economic benefit of vector databases. Why would a large corporate AI system need a stand-alone vector db like Qdrant? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether.
|
||||
People still don’t understand the economic benefit of vector databases. Why would a large corporate AI system need a standalone vector database like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether.
|
||||
|
||||
**How much would all of this cost?**
|
||||
|
||||
@@ -81,12 +83,12 @@ Julien Simon from HuggingFace says it best:
|
||||
> RAG is not a workaround for limited context size. For mission-critical enterprise use cases, RAG is a way to leverage high-value, proprietary company knowledge that will never be found in public datasets used for LLM training. At the moment, the best place to index and query this knowledge is some sort of vector index. In addition, RAG downgrades the LLM to a writing assistant. Since built-in knowledge becomes much less important, a nice small 7B open-source model usually does the trick at a fraction of the cost of a huge generic model.
|
||||
|
||||
|
||||
## Long Live RAG
|
||||
## Get superior accuracy with Qdrant's vector database
|
||||
|
||||
As LLMs continue to require enormous computing power, users will need to leverage vector search and RAG.
|
||||
As LLMs continue to require enormous computing power, users will need to leverage vector search and [RAG](https://qdrant.tech/).
|
||||
|
||||
Our customers remind us of this fact every day. As a product, our vector database is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy.
|
||||
Our customers remind us of this fact every day. As a product, [our vector database](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy.
|
||||
|
||||
We want to keep Qdrant compact, efficient and with a focused purpose. This purpose is to empower our customers to use it however they see fit.
|
||||
|
||||
When large enterprises release their generative AI into production, they need to keep costs under control, while retaining the best possible quality of responses. Qdrant has the tools to do just that. Whether through [RAG, Semantic Search, Dissimilarity Search, Recommendations or Multimodality](/articles/vector-similarity-beyond-search/) - Qdrant will continue to journey on.
|
||||
When large enterprises release their generative AI into production, they need to keep costs under control, while retaining the best possible quality of responses. Qdrant has the [vector search solutions](https://qdrant.tech/solutions/) to do just that. Revolutionize your vector search capabilities and get started with [a Qdrant demo](https://qdrant.tech/contact-us/).
|
||||
+689
@@ -0,0 +1,689 @@
|
||||
---
|
||||
title: "Optimizing RAG Through an Evaluation-Based Methodology"
|
||||
short_description: Learn how Qdrant-powered RAG applications can be tested and iteratively improved using LLM evaluation tools like Quotient.
|
||||
description: Learn how Qdrant-powered RAG applications can be tested and iteratively improved using LLM evaluation tools like Quotient.
|
||||
social_preview_image: /articles_data/rapid-rag-optimization-with-qdrant-and-quotient/preview/social_preview.jpg
|
||||
small_preview_image: /articles_data/rapid-rag-optimization-with-qdrant-and-quotient/icon.svg
|
||||
preview_dir: /articles_data/rapid-rag-optimization-with-qdrant-and-quotient/preview
|
||||
weight: -131
|
||||
author: Atita Arora
|
||||
author_link: https://github.com/atarora
|
||||
date: 2024-06-12T00:00:00.000Z
|
||||
draft: false
|
||||
keywords:
|
||||
- vector database
|
||||
- vector search
|
||||
- retrieval augmented generation
|
||||
- quotient
|
||||
- optimization
|
||||
- rag
|
||||
---
|
||||
|
||||
In today's fast-paced, information-rich world, AI is revolutionizing knowledge management. The systematic process of capturing, distributing, and effectively using knowledge within an organization is one of the fields in which AI provides exceptional value today.
|
||||
|
||||
> The potential for AI-powered knowledge management increases when leveraging Retrieval Augmented Generation (RAG), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases.
|
||||
|
||||
This process enhances the accuracy, relevance, and reliability of generated text, thereby mitigating the risk of faulty, incorrect, or nonsensical results sometimes associated with traditional LLMs. This method not only ensures that the answers are contextually relevant but also up-to-date, reflecting the latest insights and data available.
|
||||
|
||||
While RAG enhances the accuracy, relevance, and reliability of traditional LLM solutions, **an evaluation strategy can further help teams ensure their AI products meet these benchmarks of success.**
|
||||
|
||||
## Relevant tools for this experiment
|
||||
|
||||
In this article, we’ll break down a RAG Optimization workflow experiment that demonstrates that evaluation is essential to build a successful RAG strategy. We will use Qdrant and Quotient for this experiment.
|
||||
|
||||
[Qdrant](https://qdrant.tech/) is a vector database and vector similarity search engine designed for efficient storage and retrieval of high-dimensional vectors. Because Qdrant offers efficient indexing and searching capabilities, it is ideal for implementing RAG solutions, where quickly and accurately retrieving relevant information from extremely large datasets is crucial. Qdrant also offers a wealth of additional features, such as quantization, multivector support and multi-tenancy.
|
||||
|
||||
Alongside Qdrant we will use Quotient, which provides a seamless way to evaluate your RAG implementation, accelerating and improving the experimentation process.
|
||||
|
||||
[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build evaluation frameworks and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users.
|
||||
|
||||
> 💡 The [accompanying notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient) for this exercise can be found on GitHub for future reference.
|
||||
|
||||
## Summary of key findings
|
||||
|
||||
1. **Irrelevance and Hallucinations**: When the documents retrieved are irrelevant, evidenced by low scores in both Chunk Relevance and Context Relevance, the model is prone to generating inaccurate or fabricated information.
|
||||
2. **Optimizing Document Retrieval**: By retrieving a greater number of documents and reducing the chunk size, we observed improved outcomes in the model's performance.
|
||||
3. **Adaptive Retrieval Needs**: Certain queries may benefit from accessing more documents. Implementing a dynamic retrieval strategy that adjusts based on the query could enhance accuracy.
|
||||
4. **Influence of Model and Prompt Variations**: Alterations in language models or the prompts used can significantly impact the quality of the generated responses, suggesting that fine-tuning these elements could optimize performance.
|
||||
|
||||
Let us walk you through how we arrived at these findings!
|
||||
|
||||
## Building a RAG pipeline
|
||||
|
||||
To evaluate a RAG pipeline , we will have to build a RAG Pipeline first. In the interest of simplicity, we are building a Naive RAG in this article. There are certainly other versions of RAG :
|
||||
|
||||

|
||||
|
||||
The illustration below depicts how we can leverage a RAG Evaluation framework to assess the quality of RAG Application.
|
||||
|
||||

|
||||
|
||||
We are going to build a RAG application using Qdrant’s Documentation and the premeditated [hugging face dataset]([https://huggingface.co/datasets/atitaarora/qdrant_doc](https://huggingface.co/datasets/atitaarora/qdrant_doc)).
|
||||
We will then assess our RAG application’s ability to answer questions about Qdrant.
|
||||
|
||||
To prepare our knowledge store we will use Qdrant, which can be leveraged in 3 different ways as below :
|
||||
|
||||
```python
|
||||
##Uncomment to initialise qdrant client in memory
|
||||
#client = qdrant_client.QdrantClient(
|
||||
# location=":memory:",
|
||||
#)
|
||||
|
||||
##Uncomment below to connect to Qdrant Cloud
|
||||
client = qdrant_client.QdrantClient(
|
||||
os.environ.get("QDRANT_URL"),
|
||||
api_key=os.environ.get("QDRANT_API_KEY"),
|
||||
)
|
||||
|
||||
## Uncomment below to connect to local Qdrant
|
||||
#client = qdrant_client.QdrantClient("http://localhost:6333")
|
||||
```
|
||||
|
||||
We will be using [Qdrant Cloud](https://cloud.qdrant.io/login) so it is a good idea to provide the `QDRANT_URL` and `QDRANT_API_KEY` as environment variables for easier access.
|
||||
|
||||
Moving on, we will need to define the collection name as :
|
||||
|
||||
```python
|
||||
COLLECTION_NAME = "qdrant-docs-quotient"
|
||||
```
|
||||
|
||||
In this case , we may need to create different collections based on the experiments we conduct.
|
||||
|
||||
To help us provide seamless embedding creations throughout the experiment, we will use Qdrant’s native embedding provider [Fastembed]([https://qdrant.github.io/fastembed/](https://qdrant.github.io/fastembed/)) which supports [many different models]([https://qdrant.github.io/fastembed/examples/Supported_Models/](https://qdrant.github.io/fastembed/examples/Supported_Models/)) including dense as well as sparse vector models.
|
||||
|
||||
We can initialize and switch the embedding model of our choice as below :
|
||||
|
||||
```python
|
||||
## Declaring the intended Embedding Model with Fastembed
|
||||
from fastembed.embedding import TextEmbedding
|
||||
|
||||
## General Fastembed specific operations
|
||||
##Initilising embedding model
|
||||
## Using Default Model - BAAI/bge-small-en-v1.5
|
||||
embedding_model = TextEmbedding()
|
||||
|
||||
## For custom model supported by Fastembed
|
||||
#embedding_model = TextEmbedding(model_name="BAAI/bge-small-en", max_length=512)
|
||||
#embedding_model = TextEmbedding(model_name="sentence-transformers/all-MiniLM-L6-v2", max_length=384)
|
||||
|
||||
## Verify the chosen Embedding model
|
||||
embedding_model.model_name
|
||||
```
|
||||
|
||||
Before implementing RAG, we need to prepare and index our data in Qdrant.
|
||||
|
||||
This involves converting textual data into vectors using a suitable encoder (e.g., sentence transformers), and storing these vectors in Qdrant for retrieval.
|
||||
|
||||
```python
|
||||
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
||||
from langchain.docstore.document import Document as LangchainDocument
|
||||
|
||||
## Load the dataset with qdrant documentation
|
||||
dataset = load_dataset("atitaarora/qdrant_doc", split="train")
|
||||
|
||||
## Dataset to langchain document
|
||||
langchain_docs = [
|
||||
LangchainDocument(page_content=doc["text"], metadata={"source": doc["source"]})
|
||||
for doc in dataset
|
||||
]
|
||||
|
||||
len(langchain_docs)
|
||||
|
||||
#Outputs
|
||||
#240
|
||||
```
|
||||
|
||||
You can preview documents in the dataset as below :
|
||||
|
||||
```python
|
||||
## Here's an example of what a document in our dataset looks like
|
||||
print(dataset[100]['text'])
|
||||
|
||||
```
|
||||
|
||||
## Evaluation dataset
|
||||
|
||||
To measure the quality of our RAG setup, we will need a representative evaluation dataset. This dataset should contain realistic questions and the expected answers.
|
||||
|
||||
Additionally, including the expected contexts for which your RAG pipeline is designed to retrieve information would be beneficial.
|
||||
|
||||
We will be using a [prebuilt evaluation dataset](https://huggingface.co/datasets/atitaarora/qdrant_doc_qna).
|
||||
|
||||
If you are struggling to make an evaluation dataset for your use case , you can use your documents and some techniques described in this [notebook](https://github.com/qdrant/qdrant-rag-eval/blob/master/synthetic_qna/notebook/Synthetic_question_generation.ipynb)
|
||||
|
||||
### Building the RAG pipeline
|
||||
|
||||
We establish the data preprocessing parameters essential for the RAG pipeline and configure the Qdrant vector database according to the specified criteria.
|
||||
|
||||
Key parameters under consideration are:
|
||||
|
||||
- **Chunk size**
|
||||
- **Chunk overlap**
|
||||
- **Embedding model**
|
||||
- **Number of documents retrieved (retrieval window)**
|
||||
|
||||
Following the ingestion of data in Qdrant, we proceed to retrieve pertinent documents corresponding to each query. These documents are then seamlessly integrated into our evaluation dataset, enriching the contextual information within the designated **`context`** column to fulfil the evaluation aspect.
|
||||
|
||||
Next we define methods to take care of logistics with respect to adding documents to Qdrant
|
||||
|
||||
```python
|
||||
def add_documents(client, collection_name, chunk_size, chunk_overlap, embedding_model_name):
|
||||
"""
|
||||
This function adds documents to the desired Qdrant collection given the specified RAG parameters.
|
||||
"""
|
||||
|
||||
## Processing each document with desired TEXT_SPLITTER_ALGO, CHUNK_SIZE, CHUNK_OVERLAP
|
||||
text_splitter = RecursiveCharacterTextSplitter(
|
||||
chunk_size=chunk_size,
|
||||
chunk_overlap=chunk_overlap,
|
||||
add_start_index=True,
|
||||
separators=["\n\n", "\n", ".", " ", ""],
|
||||
)
|
||||
|
||||
docs_processed = []
|
||||
for doc in langchain_docs:
|
||||
docs_processed += text_splitter.split_documents([doc])
|
||||
|
||||
## Processing documents to be encoded by Fastembed
|
||||
docs_contents = []
|
||||
docs_metadatas = []
|
||||
|
||||
for doc in docs_processed:
|
||||
if hasattr(doc, 'page_content') and hasattr(doc, 'metadata'):
|
||||
docs_contents.append(doc.page_content)
|
||||
docs_metadatas.append(doc.metadata)
|
||||
else:
|
||||
# Handle the case where attributes are missing
|
||||
print("Warning: Some documents do not have 'page_content' or 'metadata' attributes.")
|
||||
|
||||
print("processed: ", len(docs_processed))
|
||||
print("content: ", len(docs_contents))
|
||||
print("metadata: ", len(docs_metadatas))
|
||||
|
||||
## Adding documents to Qdrant using desired embedding model
|
||||
client.set_model(embedding_model_name=embedding_model_name)
|
||||
client.add(collection_name=collection_name, metadata=docs_metadatas, documents=docs_contents)
|
||||
```
|
||||
|
||||
and retrieving documents from Qdrant during our RAG Pipeline assessment.
|
||||
|
||||
```python
|
||||
def get_documents(collection_name, query, num_documents=3):
|
||||
"""
|
||||
This function retrieves the desired number of documents from the Qdrant collection given a query.
|
||||
It returns a list of the retrieved documents.
|
||||
"""
|
||||
search_results = client.query(
|
||||
collection_name=collection_name,
|
||||
query_text=query,
|
||||
limit=num_documents,
|
||||
)
|
||||
results = [r.metadata["document"] for r in search_results]
|
||||
return results
|
||||
```
|
||||
|
||||
### Setting up Quotient
|
||||
|
||||
You will need an account log in, which you can get by requesting access on [Quotient's website](https://www.quotientai.co/). Once you have an account, you can create an API key by running the `quotient authenticate` CLI command.
|
||||
|
||||
<aside>
|
||||
💡 Be sure to save your API key, since it will only be displayed once (Note: you will not have to repeat this step again until your API key expires).
|
||||
|
||||
</aside>
|
||||
|
||||
**Once you have your API key, make sure to set it as an environment variable called `QUOTIENT_API_KEY`**
|
||||
|
||||
```python
|
||||
# Import QuotientAI client and connect to QuotientAI
|
||||
from quotientai.client import QuotientClient
|
||||
from quotientai.utils import show_job_progress
|
||||
|
||||
# IMPORTANT: be sure to set your API key as an environment variable called QUOTIENT_API_KEY
|
||||
# You will need this set before running the code below. You may also uncomment the following line and insert your API key:
|
||||
# os.environ['QUOTIENT_API_KEY'] = "YOUR_API_KEY"
|
||||
|
||||
quotient = QuotientClient()
|
||||
```
|
||||
|
||||
**QuotientAI** provides a seamless way to integrate *RAG evaluation* into your applications. Here, we'll see how to use it to evaluate text generated from an LLM, based on retrieved knowledge from the Qdrant vector database.
|
||||
|
||||
After retrieving the top similar documents and populating the `context` column, we can submit the evaluation dataset to Quotient and execute an evaluation job. To run a job, all you need is your evaluation dataset and a `recipe`.
|
||||
|
||||
***A recipe is a combination of a prompt template and a specified LLM.***
|
||||
|
||||
**Quotient** orchestrates the evaluation run and handles version control and asset management throughout the experimentation process.
|
||||
|
||||
***Prior to assessing our RAG solution, it's crucial to outline our optimization goals.***
|
||||
|
||||
In the context of *question-answering on Qdrant documentation*, our focus extends beyond merely providing helpful responses. Ensuring the absence of any *inaccurate or misleading information* is paramount.
|
||||
|
||||
In other words, **we want to minimize hallucinations** in the LLM outputs.
|
||||
|
||||
For our evaluation, we will be considering the following metrics, with a focus on **Faithfulness**:
|
||||
|
||||
- **Context Relevance**
|
||||
- **Chunk Relevance**
|
||||
- **Faithfulness**
|
||||
- **ROUGE-L**
|
||||
- **BERT Sentence Similarity**
|
||||
- **BERTScore**
|
||||
|
||||
### Evaluation in action
|
||||
|
||||
The function below takes an evaluation dataset as input, which in this case contains questions and their corresponding answers. It retrieves relevant documents based on the questions in the dataset and populates the context field with this information from Qdrant. The prepared dataset is then submitted to QuotientAI for evaluation for the chosen metrics. After the evaluation is complete, the function displays aggregated statistics on the evaluation metrics followed by the summarized evaluation results.
|
||||
|
||||
```python
|
||||
def run_eval(eval_df, collection_name, recipe_id, num_docs=3, path="eval_dataset_qdrant_questions.csv"):
|
||||
"""
|
||||
This function evaluates the performance of a complete RAG pipeline on a given evaluation dataset.
|
||||
|
||||
Given an evaluation dataset (containing questions and ground truth answers),
|
||||
this function retrieves relevant documents, populates the context field, and submits the dataset to QuotientAI for evaluation.
|
||||
Once the evaluation is complete, aggregated statistics on the evaluation metrics are displayed.
|
||||
|
||||
The evaluation results are returned as a pandas dataframe.
|
||||
"""
|
||||
|
||||
# Add context to each question by retrieving relevant documents
|
||||
eval_df['documents'] = eval_df.apply(lambda x: get_documents(collection_name=collection_name,
|
||||
query=x['input_text'],
|
||||
num_documents=num_docs), axis=1)
|
||||
eval_df['context'] = eval_df.apply(lambda x: "\n".join(x['documents']), axis=1)
|
||||
|
||||
# Now we'll save the eval_df to a CSV
|
||||
eval_df.to_csv(path, index=False)
|
||||
|
||||
# Upload the eval dataset to QuotientAI
|
||||
dataset = quotient.create_dataset(
|
||||
file_path=path,
|
||||
name="qdrant-questions-eval-v1",
|
||||
)
|
||||
|
||||
# Create a new task for the dataset
|
||||
task = quotient.create_task(
|
||||
dataset_id=dataset['id'],
|
||||
name='qdrant-questions-qa-v1',
|
||||
task_type='question_answering'
|
||||
)
|
||||
|
||||
# Run a job to evaluate the model
|
||||
job = quotient.create_job(
|
||||
task_id=task['id'],
|
||||
recipe_id=recipe_id,
|
||||
num_fewshot_examples=0,
|
||||
limit=500,
|
||||
metric_ids=[5, 7, 8, 11, 12, 13, 50],
|
||||
)
|
||||
|
||||
# Show the progress of the job
|
||||
show_job_progress(quotient, job['id'])
|
||||
|
||||
# Once the job is complete, we can get our results
|
||||
data = quotient.get_eval_results(job_id=job['id'])
|
||||
|
||||
# Add the results to a pandas dataframe to get statistics on performance
|
||||
df = pd.json_normalize(data, "results")
|
||||
df_stats = df[df.columns[df.columns.str.contains("metric|completion_time")]]
|
||||
|
||||
df.columns = df.columns.str.replace("metric.", "")
|
||||
df_stats.columns = df_stats.columns.str.replace("metric.", "")
|
||||
|
||||
metrics = {
|
||||
'completion_time_ms':'Completion Time (ms)',
|
||||
'chunk_relevance': 'Chunk Relevance',
|
||||
'selfcheckgpt_nli_relevance':"Context Relevance",
|
||||
'selfcheckgpt_nli':"Faithfulness",
|
||||
'rougeL_fmeasure':"ROUGE-L",
|
||||
'bert_score_f1':"BERTScore",
|
||||
'bert_sentence_similarity': "BERT Sentence Similarity",
|
||||
'completion_verbosity':"Completion Verbosity",
|
||||
'verbosity_ratio':"Verbosity Ratio",}
|
||||
|
||||
df = df.rename(columns=metrics)
|
||||
df_stats = df_stats.rename(columns=metrics)
|
||||
|
||||
display(df_stats[metrics.values()].describe())
|
||||
|
||||
return df
|
||||
|
||||
main_metrics = [
|
||||
'Context Relevance',
|
||||
'Chunk Relevance',
|
||||
'Faithfulness',
|
||||
'ROUGE-L',
|
||||
'BERT Sentence Similarity',
|
||||
'BERTScore',
|
||||
]
|
||||
```
|
||||
|
||||
## Experimentation
|
||||
|
||||
Our approach is rooted in the belief that improvement thrives in an environment of exploration and discovery. By systematically testing and tweaking various components of the RAG pipeline, we aim to incrementally enhance its capabilities and performance.
|
||||
|
||||
In the following section, we dive into the details of our experimentation process, outlining the specific experiments conducted and the insights gained.
|
||||
|
||||
### Experiment 1 - Baseline
|
||||
|
||||
Parameters
|
||||
|
||||
- **Embedding Model: `bge-small-en`**
|
||||
- **Chunk size: `512`**
|
||||
- **Chunk overlap: `64`**
|
||||
- **Number of docs retrieved (Retireval Window): `3`**
|
||||
- **LLM: `Mistral-7B-Instruct`**
|
||||
|
||||
We’ll process our documents based on configuration above and ingest them into Qdrant using `add_documents` method introduced earlier
|
||||
|
||||
```python
|
||||
#experiment1 - base config
|
||||
chunk_size = 512
|
||||
chunk_overlap = 64
|
||||
embedding_model_name = "BAAI/bge-small-en"
|
||||
num_docs = 3
|
||||
|
||||
COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}"
|
||||
|
||||
add_documents(client,
|
||||
collection_name=COLLECTION_NAME,
|
||||
chunk_size=chunk_size,
|
||||
chunk_overlap=chunk_overlap,
|
||||
embedding_model_name=embedding_model_name)
|
||||
|
||||
#Outputs
|
||||
#processed: 4504
|
||||
#content: 4504
|
||||
#metadata: 4504
|
||||
```
|
||||
|
||||
Notice the `COLLECTION_NAME` which helps us segregate and identify our collections based on the experiments conducted.
|
||||
|
||||
To proceed with the evaluation, let’s create the `evaluation recipe` up next
|
||||
|
||||
```python
|
||||
# Create a recipe for the generator model and prompt template
|
||||
recipe_mistral = quotient.create_recipe(
|
||||
model_id=10,
|
||||
prompt_template_id=1,
|
||||
name='mistral-7b-instruct-qa-with-rag',
|
||||
description='Mistral-7b-instruct using a prompt template that includes context.'
|
||||
)
|
||||
recipe_mistral
|
||||
|
||||
#Outputs recipe JSON with the used prompt template
|
||||
#'prompt_template': {'id': 1,
|
||||
# 'name': 'Default Question Answering Template',
|
||||
# 'variables': '["input_text","context"]',
|
||||
# 'created_at': '2023-12-21T22:01:54.632367',
|
||||
# 'template_string': 'Question: {input_text}\\n\\nContext: {context}\\n\\nAnswer:',
|
||||
# 'owner_profile_id': None}
|
||||
```
|
||||
|
||||
To get a list of your existing recipes, you can simply run:
|
||||
|
||||
```python
|
||||
quotient.list_recipes()
|
||||
```
|
||||
|
||||
Notice the recipe template is a simplest prompt using `Question` from evaluation template `Context` from document chunks retrieved from Qdrant and `Answer` generated by the pipeline.
|
||||
|
||||
To kick off the evaluation
|
||||
|
||||
```python
|
||||
# Kick off an evaluation job
|
||||
experiment_1 = run_eval(eval_df,
|
||||
collection_name=COLLECTION_NAME,
|
||||
recipe_id=recipe_mistral['id'],
|
||||
num_docs=num_docs,
|
||||
path=f"{COLLECTION_NAME}_{num_docs}_mistral.csv")
|
||||
```
|
||||
|
||||
This may take few minutes (depending on the size of evaluation dataset!)
|
||||
|
||||
We can look at the results from our first (baseline) experiment as below :
|
||||
|
||||

|
||||
|
||||
Notice that we have a pretty **low average Chunk Relevance** and **very large standard deviations for both Chunk Relevance and Context Relevance**.
|
||||
|
||||
Let's take a look at some of the lower performing datapoints with **poor Faithfulness**:
|
||||
|
||||
```python
|
||||
with pd.option_context('display.max_colwidth', 0):
|
||||
display(experiment_1[['content.input_text', 'content.answer','content.documents','Chunk Relevance','Context Relevance','Faithfulness']
|
||||
].sort_values(by='Faithfulness').head(2))
|
||||
```
|
||||
|
||||

|
||||
|
||||
In instances where the retrieved documents are **irrelevant (where both Chunk Relevance and Context Relevance are low)**, the model also shows **tendencies to hallucinate** and **produce poor quality responses**.
|
||||
|
||||
The quality of the retrieved text directly impacts the quality of the LLM-generated answer. Therefore, our focus will be on enhancing the RAG setup by **adjusting the chunking parameters**.
|
||||
|
||||
### Experiment 2 - Adjusting the chunk parameter
|
||||
|
||||
Keeping all other parameters constant, we changed the `chunk size` and `chunk overlap` to see if we can improve our results.
|
||||
|
||||
Parameters :
|
||||
|
||||
- **Embedding Model : `bge-small-en`**
|
||||
- **Chunk size: `1024`**
|
||||
- **Chunk overlap: `128`**
|
||||
- **Number of docs retrieved (Retireval Window): `3`**
|
||||
- **LLM: `Mistral-7B-Instruct`**
|
||||
|
||||
We will reprocess the data with the updated parameters above:
|
||||
|
||||
```python
|
||||
## for iteration 2 - lets modify chunk configuration
|
||||
## We will start with creating seperate collection to store vectors
|
||||
|
||||
chunk_size = 1024
|
||||
chunk_overlap = 128
|
||||
embedding_model_name = "BAAI/bge-small-en"
|
||||
num_docs = 3
|
||||
|
||||
COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}"
|
||||
|
||||
add_documents(client,
|
||||
collection_name=COLLECTION_NAME,
|
||||
chunk_size=chunk_size,
|
||||
chunk_overlap=chunk_overlap,
|
||||
embedding_model_name=embedding_model_name)
|
||||
|
||||
#Outputs
|
||||
#processed: 2152
|
||||
#content: 2152
|
||||
#metadata: 2152
|
||||
```
|
||||
|
||||
Followed by running evaluation :
|
||||
|
||||

|
||||
|
||||
and **comparing it with the results from Experiment 1:**
|
||||
|
||||

|
||||
|
||||
We observed slight enhancements in our LLM completion metrics (including BERT Sentence Similarity, BERTScore, ROUGE-L, and Knowledge F1) with the increase in *chunk size*. However, it's noteworthy that there was a significant decrease in *Faithfulness*, which is the primary metric we are aiming to optimize.
|
||||
|
||||
Moreover, *Context Relevance* demonstrated an increase, indicating that the RAG pipeline retrieved more relevant information required to address the query. Nonetheless, there was a considerable drop in *Chunk Relevance*, implying that a smaller portion of the retrieved documents contained pertinent information for answering the question.
|
||||
|
||||
**The correlation between the rise in Context Relevance and the decline in Chunk Relevance suggests that retrieving more documents using the smaller chunk size might yield improved results.**
|
||||
|
||||
### Experiment 3 - Increasing the number of documents retrieved (retrieval window)
|
||||
|
||||
This time, we are using the same RAG setup as `Experiment 1`, but increasing the number of retrieved documents from **3** to **5**.
|
||||
|
||||
Parameters :
|
||||
|
||||
- **Embedding Model : `bge-small-en`**
|
||||
- **Chunk size: `512`**
|
||||
- **Chunk overlap: `64`**
|
||||
- **Number of docs retrieved (Retrieval Window): `5`**
|
||||
- **LLM: : `Mistral-7B-Instruct`**
|
||||
|
||||
We can use the collection from Experiment 1 and run evaluation with modified `num_docs` parameter as :
|
||||
|
||||
```python
|
||||
#collection name from Experiment 1
|
||||
COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}"
|
||||
|
||||
#running eval for experiment 3
|
||||
experiment_3 = run_eval(eval_df,
|
||||
collection_name=COLLECTION_NAME,
|
||||
recipe_id=recipe_mistral['id'],
|
||||
num_docs=num_docs,
|
||||
path=f"{COLLECTION_NAME}_{num_docs}_mistral.csv")
|
||||
```
|
||||
|
||||
Observe the results as below :
|
||||
|
||||

|
||||
|
||||
Comparing the results with Experiment 1 and 2 :
|
||||
|
||||

|
||||
|
||||
As anticipated, employing the smaller chunk size while retrieving a larger number of documents resulted in achieving the highest levels of both *Context Relevance* and *Chunk Relevance.* Additionally, it yielded the **best** (albeit marginal) *Faithfulness* score, indicating a *reduced occurrence of inaccuracies or hallucinations*.
|
||||
|
||||
Looks like we have achieved a good hold on our chunking parameters but it is worth testing another embedding model to see if we can get better results.
|
||||
|
||||
### Experiment 4 - Changing the embedding model
|
||||
|
||||
Let us try using **MiniLM** for this experiment
|
||||
****Parameters :
|
||||
|
||||
- **Embedding Model : `MiniLM-L6-v2`**
|
||||
- **Chunk size: `512`**
|
||||
- **Chunk overlap: `64`**
|
||||
- **Number of docs retrieved (Retrieval Window): `5`**
|
||||
- **LLM: : `Mistral-7B-Instruct`**
|
||||
|
||||
We will have to create another collection for this experiment :
|
||||
|
||||
```python
|
||||
#experiment-4
|
||||
chunk_size=512
|
||||
chunk_overlap=64
|
||||
embedding_model_name="sentence-transformers/all-MiniLM-L6-v2"
|
||||
num_docs=5
|
||||
|
||||
COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}"
|
||||
|
||||
add_documents(client,
|
||||
collection_name=COLLECTION_NAME,
|
||||
chunk_size=chunk_size,
|
||||
chunk_overlap=chunk_overlap,
|
||||
embedding_model_name=embedding_model_name)
|
||||
|
||||
#Outputs
|
||||
#processed: 4504
|
||||
#content: 4504
|
||||
#metadata: 4504
|
||||
```
|
||||
|
||||
We will observe our evaluations as :
|
||||
|
||||

|
||||
|
||||
Comparing these with our previous experiments :
|
||||
|
||||

|
||||
|
||||
It appears that `bge-small` was more proficient in capturing the semantic nuances of the Qdrant Documentation.
|
||||
|
||||
Up to this point, our experimentation has focused solely on the *retrieval aspect* of our RAG pipeline. Now, let's explore altering the *generation aspect* or LLM while retaining the optimal parameters identified in Experiment 3.
|
||||
|
||||
### Experiment 5 - Changing the LLM
|
||||
|
||||
Parameters :
|
||||
|
||||
- **Embedding Model : `bge-small-en`**
|
||||
- **Chunk size: `512`**
|
||||
- **Chunk overlap: `64`**
|
||||
- **Number of docs retrieved (Retrieval Window): `5`**
|
||||
- **LLM: : `GPT-3.5-turbo`**
|
||||
|
||||
For this we can repurpose our collection from Experiment 3 while the evaluations to use a new recipe with **GPT-3.5-turbo** model.
|
||||
|
||||
```python
|
||||
#collection name from Experiment 3
|
||||
COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}"
|
||||
|
||||
# We have to create a recipe using the same prompt template and GPT-3.5-turbo
|
||||
recipe_gpt = quotient.create_recipe(
|
||||
model_id=5,
|
||||
prompt_template_id=1,
|
||||
name='gpt3.5-qa-with-rag-recipe-v1',
|
||||
description='GPT-3.5 using a prompt template that includes context.'
|
||||
)
|
||||
|
||||
recipe_gpt
|
||||
|
||||
#Outputs
|
||||
#{'id': 495,
|
||||
# 'name': 'gpt3.5-qa-with-rag-recipe-v1',
|
||||
# 'description': 'GPT-3.5 using a prompt template that includes context.',
|
||||
# 'model_id': 5,
|
||||
# 'prompt_template_id': 1,
|
||||
# 'created_at': '2024-05-03T12:14:58.779585',
|
||||
# 'owner_profile_id': 34,
|
||||
# 'system_prompt_id': None,
|
||||
# 'prompt_template': {'id': 1,
|
||||
# 'name': 'Default Question Answering Template',
|
||||
# 'variables': '["input_text","context"]',
|
||||
# 'created_at': '2023-12-21T22:01:54.632367',
|
||||
# 'template_string': 'Question: {input_text}\\n\\nContext: {context}\\n\\nAnswer:',
|
||||
# 'owner_profile_id': None},
|
||||
# 'model': {'id': 5,
|
||||
# 'name': 'gpt-3.5-turbo',
|
||||
# 'endpoint': 'https://api.openai.com/v1/chat/completions',
|
||||
# 'revision': 'placeholder',
|
||||
# 'created_at': '2024-02-06T17:01:21.408454',
|
||||
# 'model_type': 'OpenAI',
|
||||
# 'description': 'Returns a maximum of 4K output tokens.',
|
||||
# 'owner_profile_id': None,
|
||||
# 'external_model_config_id': None,
|
||||
# 'instruction_template_cls': 'NoneType'}}
|
||||
```
|
||||
|
||||
Running the evaluations as :
|
||||
|
||||
```python
|
||||
experiment_5 = run_eval(eval_df,
|
||||
collection_name=COLLECTION_NAME,
|
||||
recipe_id=recipe_gpt['id'],
|
||||
num_docs=num_docs,
|
||||
path=f"{COLLECTION_NAME}_{num_docs}_gpt.csv")
|
||||
```
|
||||
|
||||
We observe :
|
||||
|
||||

|
||||
|
||||
and comparing all the 5 experiments as below :
|
||||
|
||||

|
||||
|
||||
**GPT-3.5 surpassed Mistral-7B in all metrics**! Notably, Experiment 5 exhibited the **lowest occurrence of hallucination**.
|
||||
|
||||
## Conclusions
|
||||
|
||||
Let’s take a look at our results from all 5 experiments above
|
||||
|
||||

|
||||
|
||||
We still have a long way to go in improving the retrieval performance of RAG, as indicated by our generally poor results thus far. It might be beneficial to **explore alternative embedding models** or **different retrieval strategies** to address this issue.
|
||||
|
||||
The significant variations in *Context Relevance* suggest that **certain questions may necessitate retrieving more documents than others**. Therefore, investigating a **dynamic retrieval strategy** could be worthwhile.
|
||||
|
||||
Furthermore, there's ongoing **exploration required on the generative aspect** of RAG.
|
||||
Modifying LLMs or prompts can substantially impact the overall quality of responses.
|
||||
|
||||
This iterative process demonstrates how, starting from scratch, continual evaluation and adjustments throughout experimentation can lead to the development of an enhanced RAG system.
|
||||
|
||||
## Watch this workshop on YouTube
|
||||
|
||||
> A workshop version of this article is [available on YouTube](https://www.youtube.com/watch?v=3MEMPZR1aZA). Follow along using our [GitHub notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient).
|
||||
|
||||
<iframe width="560" height="315" src="https://www.youtube.com/embed/3MEMPZR1aZA?si=n38oTBMtH3LNCTzd" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" referrerpolicy="strict-origin-when-cross-origin" allowfullscreen></iframe>
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "Qdrant under the hood: Scalar Quantization"
|
||||
short_description: "Scalar Quantization is a newly introduced mechanism of reducing the memory footprint and increasing performance"
|
||||
description: "Scalar Quantization is a newly introduced mechanism of reducing the memory footprint and increasing performance"
|
||||
title: "Scalar Quantization: Background, Practices & More | Qdrant"
|
||||
short_description: "Discover scalar quantization for optimized data storage and improved performance, including data compression benefits and efficiency enhancements."
|
||||
description: "Discover the efficiency of scalar quantization for optimized data storage and enhanced performance. Learn about its data compression benefits and efficiency improvements."
|
||||
social_preview_image: /articles_data/scalar-quantization/social_preview.png
|
||||
small_preview_image: /articles_data/scalar-quantization/scalar-quantization-icon.svg
|
||||
preview_dir: /articles_data/scalar-quantization/preview
|
||||
@@ -15,6 +15,7 @@ keywords:
|
||||
- scalar quantization
|
||||
- memory optimization
|
||||
---
|
||||
# Efficiency Unleashed: The Power of Scalar Quantization
|
||||
|
||||
High-dimensional vector embeddings can be memory-intensive, especially when working with
|
||||
large datasets consisting of millions of vectors. Memory footprint really starts being
|
||||
@@ -38,7 +39,7 @@ from version 1.1.0, you can also optimize your memory by compressing the embeddi
|
||||
We've implemented the mechanism of **Scalar Quantization**! It turns out to have not
|
||||
only a positive impact on memory but also on the performance.
|
||||
|
||||
## Scalar Quantization
|
||||
## Scalar quantization
|
||||
|
||||
Scalar quantization is a data compression technique that converts floating point values
|
||||
into integers. In case of Qdrant `float32` gets converted into `int8`, so a single number
|
||||
@@ -254,7 +255,7 @@ In all the cases, the decrease in search precision is negligible, but we keep a
|
||||
reduction of at least 28.57%, even up to 60,64%, while searching. As a rule of thumb,
|
||||
the higher the dimensionality of the vectors, the lower the precision loss.
|
||||
|
||||
### Oversampling and Rescoring
|
||||
### Oversampling and rescoring
|
||||
|
||||
A distinctive feature of the Qdrant architecture is the ability to combine the search for quantized and original vectors in a single query.
|
||||
This enables the best combination of speed, accuracy, and RAM usage.
|
||||
@@ -286,7 +287,7 @@ The mechanism of Scalar Quantization with rescoring disabled pushes the limits o
|
||||
machines even further. It seems like handling lots of requests does not require an
|
||||
expensive setup if you can agree to a small decrease in the search precision.
|
||||
|
||||
### Good practices
|
||||
### Accessing best practices
|
||||
|
||||
Qdrant documentation on [Scalar Quantization](/documentation/quantization/#setting-up-quantization-in-qdrant)
|
||||
is a great resource describing different scenarios and strategies to achieve up to 4x
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: "Sparse Vectors in Qdrant: Pure Vector-based Hybrid Search"
|
||||
short_description: "Combining the precision of exact keyword search with NN-based ranking"
|
||||
description: "Sparse vectors are the generalization of TF-IDF and BM25, that allows to leverage the power of neural networks for text retrieval."
|
||||
title: "What is a Sparse Vector? How to Achieve Vector-based Hybrid Search"
|
||||
short_description: "Discover sparse vectors, their function, and significance in modern data processing, including methods like SPLADE for efficient use."
|
||||
description: "Learn what sparse vectors are, how they work, and their importance in modern data processing. Explore methods like SPLADE for creating and leveraging sparse vectors efficiently."
|
||||
social_preview_image: /articles_data/sparse-vectors/social_preview.png
|
||||
small_preview_image: /articles_data/sparse-vectors/sparse-vectors-icon.svg
|
||||
preview_dir: /articles_data/sparse-vectors/preview
|
||||
@@ -19,7 +19,7 @@ keywords:
|
||||
|
||||
Think of a library with a vast index card system. Each index card only has a few keywords marked out (sparse vector) of a large possible set for each book (document). This is what sparse vectors enable for text.
|
||||
|
||||
## What is a Sparse Vector?
|
||||
## What are sparse and dense vectors?
|
||||
|
||||
Sparse vectors are like the Marie Kondo of data—keeping only what sparks joy (or relevance, in this case).
|
||||
|
||||
@@ -45,7 +45,7 @@ BM25 relies solely on the frequency of words in a document and does not attempt
|
||||
Sparse vectors harness the power of neural networks to surmount these limitations while retaining the ability to query exact words and phrases.
|
||||
They excel in handling large text data, making them crucial in modern data processing a and marking an advancement over traditional methods such as BM25.
|
||||
|
||||
# Understanding Sparse Vectors
|
||||
# Understanding sparse vectors
|
||||
|
||||
Sparse Vectors are a representation where each dimension corresponds to a word or subword, greatly aiding in interpreting document rankings. This clarity is why sparse vectors are essential in modern search and recommendation systems, complimenting the meaning-rich embedding or dense vectors.
|
||||
|
||||
@@ -60,9 +60,9 @@ For example, in the medical domain, many rare terms are not present in the gener
|
||||
| **Data Representation** | Majority of elements are zero | All elements are non-zero |
|
||||
| **Computational Efficiency** | Generally higher, especially in operations involving zero elements | Lower, as operations are performed on all elements |
|
||||
| **Information Density** | Less dense, focuses on key features | Highly dense, capturing nuanced relationships |
|
||||
| **Example Applications** | Text search, Hybrid search | RAG, many general machine learning tasks |
|
||||
| **Example Applications** | Text search, Hybrid search | [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/), many general machine learning tasks |
|
||||
|
||||
Where do Sparse Vectors fail though? They're not great at capturing nuanced relationships between words. For example, they can't capture the relationship between "king" and "queen" as well as dense vectors.
|
||||
Where do sparse vectors fail though? They're not great at capturing nuanced relationships between words. For example, they can't capture the relationship between "king" and "queen" as well as dense vectors.
|
||||
|
||||
# SPLADE
|
||||
|
||||
@@ -86,13 +86,13 @@ SPLADE is quite flexible as a method, with regularization knobs that can be tune
|
||||
|
||||
First, let's look at how to create a sparse vector. Then, we'll look at the concepts behind SPLADE.
|
||||
|
||||
# Creating a Sparse Vector
|
||||
## Creating a sparse vector
|
||||
|
||||
We'll explore two different ways to create a sparse vector. The higher performance way to create a sparse vector from dedicated document and query encoders. We'll look at a simpler approach -- here we will use the same model for both document and query. We will get a dictionary of token ids and their corresponding weights for a sample text - representing a document.
|
||||
|
||||
If you'd like to follow along, here's a [Colab Notebook](https://colab.research.google.com/gist/NirantK/ad658be3abefc09b17ce29f45255e14e/splade-single-encoder.ipynb), [alternate link](https://gist.github.com/NirantK/ad658be3abefc09b17ce29f45255e14e) with all the code.
|
||||
|
||||
## Setting Up
|
||||
### Setting Up
|
||||
```python
|
||||
from transformers import AutoModelForMaskedLM, AutoTokenizer
|
||||
|
||||
@@ -104,7 +104,7 @@ model = AutoModelForMaskedLM.from_pretrained(model_id)
|
||||
text = """Arthur Robert Ashe Jr. (July 10, 1943 – February 6, 1993) was an American professional tennis player. He won three Grand Slam titles in singles and two in doubles."""
|
||||
```
|
||||
|
||||
## Computing the Sparse Vector
|
||||
### Computing the sparse vector
|
||||
```python
|
||||
import torch
|
||||
|
||||
@@ -130,7 +130,7 @@ print(vec.shape)
|
||||
|
||||
You'll notice that there are 38 tokens in the text based on this tokenizer. This will be different from the number of tokens in the vector. In a TF-IDF, we'd assign weights only to these tokens or words. In SPLADE, we assign weights to all the tokens in the vocabulary using this vector using our learned model.
|
||||
|
||||
# Term Expansion and Weights
|
||||
## Term expansion and weights
|
||||
```python
|
||||
def extract_and_map_sparse_vector(vector, tokenizer):
|
||||
"""
|
||||
@@ -202,7 +202,7 @@ If you're interested in using the higher-performance approach, check out the fol
|
||||
1. [naver/efficient-splade-VI-BT-large-doc](https://huggingface.co/naver/efficient-splade-vi-bt-large-doc)
|
||||
2. [naver/efficient-splade-VI-BT-large-query](https://huggingface.co/naver/efficient-splade-vi-bt-large-doc)
|
||||
|
||||
## Why SPLADE works? Term Expansion
|
||||
## Why SPLADE works: term expansion
|
||||
|
||||
Consider a query "solar energy advantages". SPLADE might expand this to include terms like "renewable," "sustainable," and "photovoltaic," which are contextually relevant but not explicitly mentioned. This process is called term expansion, and it's a key component of SPLADE.
|
||||
|
||||
@@ -218,7 +218,7 @@ For example, assume a 1M document corpus. Say, we use 100 sparse token ids + wei
|
||||
| OpenAI Embedding | 12.288 |
|
||||
| Sparse Vector | 1.12 |
|
||||
|
||||
## How SPLADE works? Leveraging BERT
|
||||
## How SPLADE works: leveraging BERT
|
||||
|
||||
SPLADE leverages a transformer architecture to generate sparse representations of documents and queries, enabling efficient retrieval. Let's dive into the process.
|
||||
|
||||
@@ -239,15 +239,15 @@ A downside of dense vectors is that they are not interpretable, making it diffic
|
||||
|
||||
SPLADE importance estimation can provide insights into the 'why' behind a document's relevance to a query. By shedding light on which tokens contribute most to the retrieval score, SPLADE offers some degree of interpretability alongside performance, a rare feat in the realm of neural IR systems. For engineers working on search, this transparency is invaluable.
|
||||
|
||||
## Known Limitations of SPLADE
|
||||
## Known limitations of SPLADE
|
||||
|
||||
### Pooling Strategy
|
||||
### Pooling strategy
|
||||
The switch to max pooling in SPLADE improved its performance on the MS MARCO and TREC datasets. However, this indicates a potential limitation of the baseline SPLADE pooling method, suggesting that SPLADE's performance is sensitive to the choice of pooling strategy.
|
||||
|
||||
### Document and Query Encoder
|
||||
### Document and query Eecoder
|
||||
The SPLADE model variant that uses a document encoder with max pooling but no query encoder reaches the same performance level as the prior SPLADE model. This suggests a limitation in the necessity of a query encoder, potentially affecting the efficiency of the model.
|
||||
|
||||
## Other Sparse Vector Methods
|
||||
## Other sparse vector methods
|
||||
|
||||
SPLADE is not the only method to create sparse vectors.
|
||||
|
||||
@@ -259,8 +259,7 @@ This method preserves the ability to query exact words and phrases but avoids th
|
||||
|
||||
We will cover these methods in detail in a future article.
|
||||
|
||||
|
||||
# Leveraging Sparse Vectors in Qdrant for Hybrid Search
|
||||
## Leveraging sparse vectors in Qdrant for hybrid search
|
||||
|
||||
Qdrant supports a separate index for Sparse Vectors.
|
||||
This enables you to use the same collection for both dense and sparse vectors.
|
||||
@@ -268,7 +267,7 @@ Each "Point" in Qdrant can have both dense and sparse vectors.
|
||||
|
||||
But let's first take a look at how you can work with sparse vectors in Qdrant.
|
||||
|
||||
## Practical Implementation in Python
|
||||
## Practical implementation in Python
|
||||
|
||||
Let's dive into how Qdrant handles sparse vectors with an example. Here is what we will cover:
|
||||
|
||||
@@ -282,7 +281,7 @@ Let's dive into how Qdrant handles sparse vectors with an example. Here is what
|
||||
|
||||
5. Retrieving and Interpreting Results: The search operation returns results that include the id of the matching document, its score, and other relevant details. The score is a crucial aspect, reflecting the similarity between the query and the documents in the collection.
|
||||
|
||||
### 1. Setting up
|
||||
### 1. Set up
|
||||
|
||||
```python
|
||||
# Qdrant client setup
|
||||
@@ -295,7 +294,7 @@ COLLECTION_NAME = "example_collection"
|
||||
point_id = 1 # Assign a unique ID for the point
|
||||
```
|
||||
|
||||
### 2. Creating a Collection with Sparse Vector Support
|
||||
### 2. Create a collection with sparse vector support
|
||||
|
||||
```python
|
||||
client.recreate_collection(
|
||||
@@ -312,7 +311,7 @@ client.recreate_collection(
|
||||
```
|
||||
|
||||
|
||||
### 3. Inserting Sparse Vectors
|
||||
### 3. Insert sparse vectors
|
||||
|
||||
Here, we see the process of inserting a sparse vector into the Qdrant collection. This step is key to building a dataset that can be quickly retrieved in the first stage of the retrieval process, utilizing the efficiency of sparse vectors. Since this is for demonstration purposes, we insert only one point with Sparse Vector and no dense vector.
|
||||
|
||||
@@ -336,7 +335,7 @@ By upserting points with sparse vectors, we prepare our dataset for rapid first-
|
||||
|
||||
Those familiar with the Qdrant API will notice that the extra care taken to be consistent with the existing named vectors API -- this is to make it easier to use sparse vectors in existing codebases. As always, you're able to **apply payload filters**, shard keys, and other advanced features you've come to expect from Qdrant. To make things easier for you, the indices and values don't have to be sorted before upsert. Qdrant will sort them when the index is persisted e.g. on disk.
|
||||
|
||||
### 4. Querying with Sparse Vectors
|
||||
### 4. Query with sparse vectors
|
||||
|
||||
We use the same process to prepare a query vector as well. This involves computing the vector from a query text and extracting its indices and values. We then use these details to construct a query against our collection.
|
||||
|
||||
@@ -353,7 +352,7 @@ query_values = query_vec.detach().numpy()[indices]
|
||||
|
||||
In this example, we use the same model for both document and query. This is not a requirement, but it's a simpler approach.
|
||||
|
||||
### 5. Retrieving and Interpreting Results
|
||||
### 5. Retrieve and interpret results
|
||||
|
||||
After setting up the collection and inserting sparse vectors, the next critical step is retrieving and interpreting the results. This process involves executing a search query and then analyzing the returned results.
|
||||
|
||||
@@ -406,7 +405,7 @@ $$\text{Similarity}(\text{Query}, \text{Document}) = \sum_{i \in I} \text{Query}
|
||||
This formula calculates the similarity score by multiplying corresponding elements of the query and document vectors and summing these products. This method is particularly effective with sparse vectors, where many elements are zero, leading to a computationally efficient process. The higher the score, the greater the similarity between the query and the document, making it a valuable metric for assessing the relevance of the retrieved documents.
|
||||
|
||||
|
||||
## Hybrid Search: Combining Sparse and Dense Vectors
|
||||
## Hybrid search: combining sparse and dense vectors
|
||||
|
||||
By combining search results from both dense and sparse vectors, you can achieve a hybrid search that is both efficient and accurate.
|
||||
Results from sparse vectors will guarantee, that all results with the required keywords are returned,
|
||||
@@ -476,7 +475,7 @@ The result will be a pair of result lists, one for dense and one for sparse vect
|
||||
|
||||
Having those results, there are several ways to combine them:
|
||||
|
||||
### Mixing or Fusion
|
||||
### Mixing or fusion
|
||||
|
||||
You can mix the results from both dense and sparse vectors, based purely on their relative scores. This is a simple and effective approach, but it doesn't take into account the semantic similarity between the results. Among the [popular mixing methods](https://medium.com/plain-simple-software/distribution-based-score-fusion-dbsf-a-new-approach-to-vector-search-ranking-f87c37488b18) are:
|
||||
|
||||
@@ -495,7 +494,7 @@ You can use obtained results as a first stage of a two-stage retrieval process.
|
||||
|
||||
And that's it! You've successfully achieved hybrid search with Qdrant!
|
||||
|
||||
## Additional Resources
|
||||
## Additional resources
|
||||
For those who want to dive deeper, here are the top papers on the topic most of which have code available:
|
||||
|
||||
1. Problem Motivation: [Sparse Overcomplete Word Vector Representations](https://ar5iv.org/abs/1506.02004?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors)
|
||||
@@ -504,7 +503,7 @@ For those who want to dive deeper, here are the top papers on the topic most of
|
||||
1. Late Interaction - [ColBERTv2: Effective and Efficient Retrieval via Lightweight Late Interaction](https://ar5iv.org/abs/2112.01488?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors)
|
||||
1. [SparseEmbed: Learning Sparse Lexical Representations with Contextual Embeddings for Retrieval](https://research.google/pubs/pub52289/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors)
|
||||
|
||||
**Why just read when you try it out?**
|
||||
**Why just read when you can try it out?**
|
||||
|
||||
We've packed an easy-to-use Colab for you on how to make a Sparse Vector: [Sparse Vectors Single Encoder Demo](https://colab.research.google.com/drive/1wa2Yr5BCOgV0MTOFFTude99BOXCLHXky?usp=sharing). Run it, tinker with it, and start seeing the magic unfold in your projects. We can't wait to hear how you use it!
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: "Qdrant x.y.0 - <include headline> #required; update version and headline"
|
||||
draft: true # Change to false to publish the article at https://qdrant.tech/articles/
|
||||
draft: true # Change to false to publish the article at /articles/
|
||||
slug: qdrant-x.y.z # required; subtitute version number
|
||||
short_description: "Headline-like description."
|
||||
description: "Headline with more detail. Suggested limit: 140 characters. "
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
---
|
||||
title: Vector Similarity beyond Search
|
||||
short_description: Harnessing the full capabilities of vector embeddings
|
||||
description: We explore some of the promising new techniques that can be used to expand use-cases of unstructured data and unlock new similarities-based data exploration tools.
|
||||
title: "Vector Similarity: Going Beyond Full-Text Search | Qdrant"
|
||||
short_description: Explore how vector similarity enhances data discovery beyond full-text search, including diversity sampling and more!
|
||||
description: Discover how vector similarity expands data exploration beyond full-text search. Explore diversity sampling and more for enhanced data discovery!
|
||||
preview_dir: /articles_data/vector-similarity-beyond-search/preview
|
||||
small_preview_image: /articles_data/vector-similarity-beyond-search/icon.svg
|
||||
social_preview_image: /articles_data/vector-similarity-beyond-search/preview/social_preview.jpg
|
||||
@@ -19,17 +19,23 @@ keywords:
|
||||
- recommendation
|
||||
---
|
||||
|
||||
# Vector Similarity: Unleashing Data Insights Beyond Traditional Search
|
||||
|
||||
When making use of unstructured data, there are traditional go-to solutions that are well-known for developers:
|
||||
|
||||
- **Full-text search** when you need to find documents that contain a particular word or phrase.
|
||||
- **Vector search** when you need to find documents that are semantically similar to a given query.
|
||||
- **[Vector search](https://qdrant.tech/documentation/overview/vector-search/)** when you need to find documents that are semantically similar to a given query.
|
||||
|
||||
Sometimes people mix those two approaches, so it might look like the vector similarity is just an extension of full-text search. However, in this article, we will explore some promising new techniques that can be used to expand the use-case of unstructured data and demonstrate that vector similarity creates its own stack of data exploration tools.
|
||||
|
||||
## What is vector similarity search?
|
||||
|
||||
{{< figure width=70% src=/articles_data/vector-similarity-beyond-search/venn-diagram.png caption="Full-text search and Vector Similarity Functionality overlap" >}}
|
||||
Vector similarity offers a range of powerful functions that go far beyond those available in traditional full-text search engines. From dissimilarity search to diversity and recommendation, these methods can expand the cases in which vectors are useful.
|
||||
|
||||
Vector Databases, which are designed to store and process immense amounts of vectors, are the first candidates to implement these new techniques and allow users to exploit their data to its fullest.
|
||||
|
||||
|
||||
## Vector similarity search vs. full-text search
|
||||
|
||||
While there is an intersection in the functionality of these two approaches, there is also a vast area of functions that is unique to each of them.
|
||||
For example, the exact phrase matching and counting of results are native to full-text search, while vector similarity support for this type of operation is limited.
|
||||
@@ -39,19 +45,21 @@ This mismatch in expectations might sometimes lead to confusion.
|
||||
Attempting to use a vector similarity as a full-text search can result in a range of frustrations, from slow response times to poor search results, to limited functionality.
|
||||
As an outcome, they are getting only a fraction of the benefits of vector similarity.
|
||||
|
||||
Below we will explore why vector similarity stack deserves new interfaces and design patterns that will unlock the full potential of this technology, which can still be used in conjunction with full-text search.
|
||||
{{< figure width=70% src=/articles_data/vector-similarity-beyond-search/venn-diagram.png caption="Full-text search and Vector Similarity Functionality overlap" >}}
|
||||
|
||||
Below we will explore why the vector similarity stack deserves new interfaces and design patterns that will unlock the full potential of this technology, which can still be used in conjunction with full-text search.
|
||||
|
||||
|
||||
## New Ways to Interact with Similarities
|
||||
## New ways to interact with similarities
|
||||
|
||||
Having a vector representation of unstructured data unlocks new ways of interacting with it.
|
||||
For example, it can be used to measure semantic similarity between words, to cluster words or documents based on their meaning, to find related images, or even to generate new text.
|
||||
However, these interactions can go beyond finding their nearest neighbors (kNN).
|
||||
|
||||
There are several other techniques that can be leveraged by vector representations beyond the traditional kNN search. These include dissimilarity search, diversity search, recommendations and discovery functions.
|
||||
There are several other techniques that can be leveraged by vector representations beyond the traditional kNN search. These include dissimilarity search, diversity search, recommendations, and discovery functions.
|
||||
|
||||
|
||||
## Dissimilarity Search
|
||||
## Dissimilarity ssearch
|
||||
|
||||
The Dissimilarity —or farthest— search is the most straightforward concept after the nearest search, which can’t be reproduced in a traditional full-text search.
|
||||
It aims to find the most un-similar or distant documents across the collection.
|
||||
@@ -66,21 +74,21 @@ With vector similarity, we can easily achieve a dissimilarity search by invertin
|
||||
The dissimilarity search can find items in areas where previously no other search could be used.
|
||||
Let’s look at a few examples.
|
||||
|
||||
### Case: Mislabeling Detection
|
||||
### Case: mislabeling detection
|
||||
|
||||
For example, we have a dataset of furniture in which we have classified our items into what kind of furniture they are: tables, chairs, lamps, etc.
|
||||
To ensure our catalog is accurate, we can use a dissimilarity search to highlight items that are most likely mislabeled.
|
||||
|
||||
To do this, we only need to search for the most dissimilar items using the
|
||||
embedding of the category title itself as a query.
|
||||
This can be too broad, so, combining it with filters —a [Qdrant superpower](/articles/filtrable-hnsw/)—, we can narrow down the search to a specific category.
|
||||
This can be too broad, so, by combining it with filters —a [Qdrant superpower](/articles/filtrable-hnsw/)—, we can narrow down the search to a specific category.
|
||||
|
||||
|
||||
{{< figure src=/articles_data/vector-similarity-beyond-search/mislabelling.png caption="Mislabeling Detection" >}}
|
||||
|
||||
The output of this search can be further processed with heavier models or human supervision to detect actual mislabeling.
|
||||
|
||||
### Case: Outlier Detection
|
||||
### Case: outlier detection
|
||||
|
||||
In some cases, we might not even have labels, but it is still possible to try to detect anomalies in our dataset.
|
||||
Dissimilarity search can be used for this purpose as well.
|
||||
@@ -91,7 +99,7 @@ The only thing we need is a bunch of reference points that we consider "normal".
|
||||
Then we can search for the most dissimilar points to this reference set and use them as candidates for further analysis.
|
||||
|
||||
|
||||
## Diversity Search
|
||||
## Diversity search
|
||||
|
||||
Even with no input provided vector, (dis-)similarity can improve an overall selection of items from the dataset.
|
||||
|
||||
@@ -118,7 +126,7 @@ However, there is still room for new ideas, particularly regarding diversity ret
|
||||
By utilizing more advanced vector-native engines, it could be possible to take use cases to the next level and achieve even better results.
|
||||
|
||||
|
||||
## Recommendations
|
||||
## Vector similarity recommendations
|
||||
|
||||
Vector similarity can go above a single query vector.
|
||||
It can combine multiple positive and negative examples for a more accurate retrieval.
|
||||
@@ -127,7 +135,7 @@ Doing this, we can skip query-time neural network inference, and make the recomm
|
||||
|
||||
There are multiple ways to implement recommendations with vectors.
|
||||
|
||||
### Vector-Features Recommendations
|
||||
### Vector-features recommendations
|
||||
|
||||
The first approach is to take all positive and negative examples and average them to create a single query vector.
|
||||
In this technique, the more significant components of positive vectors are canceled out by the negative ones, and the resulting vector is a combination of all the features present in the positive examples, but not in the negative ones.
|
||||
@@ -136,7 +144,7 @@ In this technique, the more significant components of positive vectors are cance
|
||||
|
||||
This approach is already implemented in Qdrant, and while it works great when the vectors are assumed to have each of their dimensions represent some kind of feature of the data, sometimes distances are a better tool to judge negative and positive examples.
|
||||
|
||||
### Relative Distance Recommendations
|
||||
### Relative distance recommendations
|
||||
|
||||
Another approach is to use the distance between negative examples to the candidates to help them create exclusion areas.
|
||||
In this technique, we perform searches near the positive examples while excluding the points that are closer to a negative example than to a positive one.
|
||||
@@ -164,37 +172,37 @@ Given a trained model, the user can provide positive and negative examples, and
|
||||
{{< figure width=60% src=/articles_data/vector-similarity-beyond-search/discovery.png caption="Reversed triplet loss" >}}
|
||||
|
||||
Multiple positive-negative pairs can be provided to make the discovery process more accurate.
|
||||
Worth mentioning, that as well as in NN training, the dataset may contain noise and some portion of contradictory information, so a discovery process should be tolerant to this kind of data imperfections.
|
||||
Worth mentioning, that as well as in NN training, the dataset may contain noise and some portion of contradictory information, so a discovery process should be tolerant of this kind of data imperfections.
|
||||
|
||||
|
||||
<!-- Image with multiple pairs -->
|
||||
{{< figure width=80% src=/articles_data/vector-similarity-beyond-search/discovery-noise.png caption="Sample pairs" >}}
|
||||
|
||||
The important difference between this and recommendation method is that the positive-negative pairs in discovery method doesn’t assume that the final result should be close to positive, it only assumes that it should be closer than the negative one.
|
||||
The important difference between this and the recommendation method is that the positive-negative pairs in the discovery method don’t assume that the final result should be close to positive, it only assumes that it should be closer than the negative one.
|
||||
|
||||
{{< figure width=80% src=/articles_data/vector-similarity-beyond-search/discovery-vs-recommendations.png caption="Discovery vs Recommendation" >}}
|
||||
|
||||
In combination with filtering or similarity search, the additional context information provided by the discovery pairs can be used as a re-ranking factor.
|
||||
|
||||
## A New API Stack for Vector Databases
|
||||
## A new API stack for vector databases
|
||||
|
||||
When you introduce vector similarity capabilities into your text search engine, you extend its functionality.
|
||||
However, it doesn't work the other way around, as the vector similarity as a concept is much broader than some task-specific implementations of full-text search.
|
||||
|
||||
Vector Databases, which introduce built-in full-text functionality, must make several compromises:
|
||||
[Vector databases](https://qdrant.tech/), which introduce built-in full-text functionality, must make several compromises:
|
||||
|
||||
- Choose a specific full-text search variant.
|
||||
- Either sacrifice API consistency or limit vector similarity functionality to only basic kNN search.
|
||||
- Introduce additional complexity to the system.
|
||||
|
||||
|
||||
Qdrant, on the contrary, puts vector similarity in the center of it's API and architecture, such that it allows us to move towards a new stack of vector-native operations.
|
||||
Qdrant, on the contrary, puts vector similarity in the center of its API and architecture, such that it allows us to move towards a new stack of vector-native operations.
|
||||
We believe that this is the future of vector databases, and we are excited to see what new use-cases will be unlocked by these techniques.
|
||||
|
||||
## Key takeaways:
|
||||
|
||||
## Wrapping up
|
||||
- Vector similarity offers advanced data exploration tools beyond traditional full-text search, including dissimilarity search, diversity sampling, and recommendation systems.
|
||||
- Practical applications of vector similarity include improving data quality through mislabeling detection and anomaly identification.
|
||||
- Enhanced user experiences are achieved by leveraging advanced search techniques, providing users with intuitive data exploration, and improving decision-making processes.
|
||||
|
||||
Vector similarity offers a range of powerful functions that go far beyond those available in traditional full-text search engines.
|
||||
From dissimilarity search to diversity and recommendation, these methods can expand the cases in which vectors are useful.
|
||||
Ready to unlock the full potential of your data? [Try a free demo](https://qdrant.tech/contact-us/) to explore how vector similarity can revolutionize your data insights and drive smarter decision-making.
|
||||
|
||||
Vector Databases, which are designed to store and process immense amounts of vectors, are the first candidates to implement these new techniques and allow users to exploit their data to its fullest.
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
---
|
||||
title: "What are Vector Embeddings?"
|
||||
title: "What are Vector Embeddings? - Revolutionize Your Search Experience"
|
||||
draft: false
|
||||
slug: what-are-embeddings?
|
||||
short_description: What are Vector Embeddings?
|
||||
description: Explore the key functionalities of vector embeddings and learn how they convert complex data into a format that machines can understand.
|
||||
short_description: Explore the power of vector embeddings. Learn to use numerical machine learning representations to build a personalized Neural Search Service with Fastembed.
|
||||
description: Discover the power of vector embeddings. Learn how to harness the potential of numerical machine learning representations to create a personalized Neural Search Service with FastEmbed.
|
||||
preview_dir: /articles_data/what-are-embeddings/preview
|
||||
weight: -102
|
||||
social_preview_image: /articles_data/what-are-embeddings/preview/social-preview.jpg
|
||||
@@ -40,7 +40,7 @@ The same embeddings can be repurposed for search, ads, and other features, creat
|
||||
They make [high-dimensional](https://www.sciencedirect.com/topics/computer-science/high-dimensional-data) data more manageable. This reduces storage requirements, improves computational efficiency, and makes sense of a ton of **unstructured** data.
|
||||
|
||||
|
||||
## Why Use Vector Embeddings?
|
||||
## Why use vector embeddings?
|
||||
|
||||
The **nuances** of natural language or the hidden **meaning** in large datasets of images, sounds, or user interactions are hard to fit into a table. Traditional relational databases can't efficiently query most types of data being currently used and produced, making the **retrieval** of this information very limited.
|
||||
|
||||
@@ -70,7 +70,7 @@ The meaning of a data point is implicitly defined by its **position** on the vec
|
||||
> The quality of the vector representations drives the performance. The embedding model that works best for you depends on your use case.
|
||||
|
||||
|
||||
### Creating Vector Embeddings
|
||||
### Creating vector embeddings
|
||||
|
||||
Embeddings translate the complexities of human language to a format that computers can understand. It uses neural networks to assign **numerical values** to the input data, in a way that similar data has similar values.
|
||||
|
||||
@@ -128,7 +128,7 @@ And then it compares contexts to known architectural and design principles:
|
||||
The model creates a vector embedding for "biophilic design" that encapsulates the concept of integrating natural elements into man-made environments. Augmented with attributes that highlight the correlation between this integration and its positive impact on health, well-being, and environmental sustainability.
|
||||
|
||||
|
||||
### Integration with Embedding APIs
|
||||
### Integration with embedding APIs
|
||||
|
||||
Selecting the right embedding model for your use case is crucial to your application performance. Qdrant makes it easier by offering seamless integration with the best selection of embedding APIs, including [Cohere](/documentation/embeddings/cohere/), [Gemini](/documentation/embeddings/gemini/), [Jina Embeddings](/documentation/embeddings/jina-embeddings/), [OpenAI](/documentation/embeddings/openai/), [Aleph Alpha](/documentation/embeddings/aleph-alpha/), [Fastembed](https://github.com/qdrant/fastembed), and [AWS Bedrock](/documentation/embeddings/bedrock/).
|
||||
|
||||
@@ -138,7 +138,7 @@ Fastembed, which we’ll use on the example below, is designed for efficiency an
|
||||
|
||||
We plan to go deeper into selecting the best model based on performance, cost, integration ease, and scalability in a future post.
|
||||
|
||||
## Create a Neural Search Service with Fastembed
|
||||
## Create a neural search service with Fastmbed
|
||||
|
||||
Now that you’re familiar with the core concepts around vector embeddings, how about start building your own [Neural Search Service](/documentation/tutorials/neural-search/)?
|
||||
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
title: "What is a Vector Database?"
|
||||
draft: false
|
||||
slug: what-is-a-vector-database?
|
||||
short_description: What is a Vector Database?
|
||||
description: An overview of vector databases, detailing their functionalities, architecture, and diverse use cases in modern data processing.
|
||||
short_description: What is a Vector Database? Use Cases & Examples | Qdrant
|
||||
description: Discover what a vector database is, its core functionalities, and real-world applications. Unlock advanced data management with our comprehensive guide.
|
||||
preview_dir: /articles_data/what-is-a-vector-database/preview
|
||||
weight: -100
|
||||
social_preview_image: /articles_data/what-is-a-vector-database/preview/social-preview.jpg
|
||||
@@ -19,8 +19,13 @@ tags:
|
||||
aliases: [ /blog/what-is-a-vector-database/ ]
|
||||
---
|
||||
|
||||
> A Vector Database is a specialized database system designed for efficiently indexing, querying, and retrieving high-dimensional vector data. Those systems enable advanced data analysis and similarity-search operations that extend well beyond the traditional, structured query approach of conventional databases.
|
||||
# Why use a Vector Database & How Does it Work?
|
||||
|
||||
In the ever-evolving landscape of data management and artificial intelligence, [vector databases](https://qdrant.tech/qdrant-vector-database/) have emerged as a revolutionary tool for efficiently handling complex, high-dimensional data. But what exactly is a vector database? This comprehensive guide delves into the fundamentals of vector databases, exploring their unique capabilities, core functionalities, and real-world applications.
|
||||
|
||||
## What is a Vector Database?
|
||||
|
||||
A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized database system designed for efficiently indexing, querying, and retrieving high-dimensional vector data. Those systems enable advanced data analysis and similarity-search operations that extend well beyond the traditional, structured query approach of conventional databases.
|
||||
|
||||
## Why use a Vector Database?
|
||||
|
||||
@@ -65,7 +70,7 @@ The **creation** of vector data (so we can store this high-dimensional data on o
|
||||
|
||||
### How do Embeddings Work?
|
||||
|
||||
Embeddings translate this high-dimensional data into a more manageable, **lower-dimensional** vector form that's more suitable for machine learning and data processing applications, typically through **neural network models**.
|
||||
[Embeddings](https://qdrant.tech/articles/what-are-embeddings/) translate this high-dimensional data into a more manageable, **lower-dimensional** vector form that's more suitable for machine learning and data processing applications, typically through **neural network models**.
|
||||
|
||||
In creating dimensions for text, for example, the process involves analyzing the text to capture its linguistic elements.
|
||||
|
||||
@@ -79,10 +84,9 @@ Each layer extracts different levels of features, such as context, semantics, an
|
||||
The final layers of the network condense this information into a vector that is a compact, lower-dimensional representation of the image but still retains the essential information.
|
||||
|
||||
|
||||
## Core Functionalities of Vector Databases
|
||||
## The Core Functionalities of Vector Databases
|
||||
|
||||
|
||||
### What is Indexing?
|
||||
### Vector Database Indexing
|
||||
|
||||
Have you ever tried to find a specific face in a massive crowd photo? Well, vector databases face a similar challenge when dealing with tons of high-dimensional vectors.
|
||||
|
||||
@@ -97,7 +101,7 @@ This way, finding similar images becomes a quick hop across related groups, inst
|
||||
Different indexing methods exist, each with its strengths. [HNSW](/articles/filtrable-hnsw/) balances speed and accuracy like a well-connected network of shortcuts in the crowd. Others, like IVF or Product Quantization, focus on specific tasks or memory efficiency.
|
||||
|
||||
|
||||
#### What is Binary Quantization?
|
||||
### Binary Quantization
|
||||
|
||||
Quantization is a technique used for reducing the total size of the database. It works by compressing vectors into a more compact representation at the cost of accuracy.
|
||||
|
||||
@@ -113,7 +117,7 @@ Think of each data point as a ruler. Binary quantization splits this ruler in ha
|
||||
This "quantized" code is much smaller and easier to compare. Especially for OpenAI embeddings, this type of quantization has proven to achieve a massive performance improvement at a lower cost of accuracy.
|
||||
|
||||
|
||||
### What is Similarity Search?
|
||||
### Similarity Search
|
||||
|
||||
[Similarity search](/documentation/concepts/search/) allows you to search not by keywords but by meaning. This way you can do searches such as similar songs that evoke the same mood, finding images that match your artistic vision, or even exploring emotional patterns in text.
|
||||
|
||||
@@ -130,9 +134,9 @@ Once the closest vectors are identified at the bottom layer, these points transl
|
||||
|
||||
### Scalability
|
||||
|
||||
Vector databases often deal with datasets that comprise billions of high-dimensional vectors. This data isn't just large in volume but also complex in nature, requiring more computing power and memory to process. Scalable systems can handle this increased complexity without performance degradation. This is achieved through a combination of a **distributed architecture**, **dynamic resource allocation**, **data partitioning**, **load balancing**, and **optimization techniques**.
|
||||
[Vector databases](https://qdrant.tech/qdrant-vector-database/) often deal with datasets that comprise billions of high-dimensional vectors. This data isn't just large in volume but also complex in nature, requiring more computing power and memory to process. Scalable systems can handle this increased complexity without performance degradation. This is achieved through a combination of a **distributed architecture**, **dynamic resource allocation**, **data partitioning**, **load balancing**, and **optimization techniques**.
|
||||
|
||||
Systems like Qdrant exemplify scalability in vector databases. It leverages Rust's efficiency in **memory management** and **performance**, which allows handling of large-scale data with optimized resource usage.
|
||||
Systems like Qdrant exemplify scalability in vector databases. It [leverages Rust's efficiency](https://qdrant.tech/articles/why-rust/) in **memory management** and **performance**, which allows the handling of large-scale data with optimized resource usage.
|
||||
|
||||
|
||||
### Efficient Query Processing
|
||||
@@ -170,7 +174,7 @@ At Qdrant, this includes mechanisms such as:
|
||||
- Advanced database monitoring and anomaly detection
|
||||
|
||||
|
||||
## Architecture of a Vector Database
|
||||
## What is the Architecture of a Vector Database?
|
||||
|
||||
A vector database is made of multiple different entities and relations. Here's a high-level overview of Qdrant's terminologies and how they fit into the larger picture:
|
||||
|
||||
@@ -191,15 +195,17 @@ Alternatively, the Memmap storage option creates a virtual address space linked
|
||||
**Clients**: Qdrant supports various programming languages for client interaction, such as Python, Go, Rust, and Typescript. This way developers can connect to and interact with Qdrant using the programming language they prefer.
|
||||
|
||||
|
||||
### Vector Database Use Cases
|
||||
## Vector Database Use Cases
|
||||
|
||||
If we had to summarize the use cases for vector databases into a single word, it would be "match". They are great at finding non-obvious ways to correspond or “match” data with a given query. Whether it's through similarity in images, text, user preferences, or patterns in data.
|
||||
If we had to summarize the [use cases for vector databases](https://qdrant.tech/use-cases/) into a single word, it would be "match". They are great at finding non-obvious ways to correspond or “match” data with a given query. Whether it's through similarity in images, text, user preferences, or patterns in data.
|
||||
|
||||
Here’s some examples on how to take advantage of using vector databases:
|
||||
Here are some examples of how to take advantage of using vector databases:
|
||||
|
||||
**Personalized recommendation systems** to analyze and interpret complex user data, such as preferences, behaviors, and interactions. For example, on Spotify, if a user frequently listens to the same song or skips it, the recommendation engine takes note of this to personalize future suggestions.
|
||||
[Personalized recommendation systems](https://qdrant.tech/recommendations/) to analyze and interpret complex user data, such as preferences, behaviors, and interactions. For example, on Spotify, if a user frequently listens to the same song or skips it, the recommendation engine takes note of this to personalize future suggestions.
|
||||
|
||||
**Semantic search** allows for systems to be able to capture the deeper semantic meaning of words and text. In modern search engines, if someone searches for "tips for planting in spring," it tries to understand the intent and contextual meaning behind the query. It doesn’t try just matching the words themselves. Here’s an example of a [vector search engine for Startups](https://demo.qdrant.tech/) made with Qdrant:
|
||||
[Semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) allows for systems to be able to capture the deeper semantic meaning of words and text. In modern search engines, if someone searches for "tips for planting in spring," it tries to understand the intent and contextual meaning behind the query. It doesn’t try just matching the words themselves.
|
||||
|
||||
Here’s an example of a [vector search engine for Startups](https://demo.qdrant.tech/) made with Qdrant:
|
||||
|
||||
|
||||

|
||||
@@ -209,9 +215,9 @@ There are many other use cases like for **fraud detection and anomaly analysis**
|
||||
Those are just a few examples. The ability of vector databases to “match” data with queries makes them essential for multiple types of applications. Here are some more [use cases examples](/use-cases/) you can take a look at.
|
||||
|
||||
|
||||
### Starting Your First Vector Database Project
|
||||
### Get Started With Qdrant’s Vector Database Today
|
||||
|
||||
Now that you're familiar with the core concepts around vector databases, it’s time to get our hands dirty. [Start by building your own semantic search engine](/documentation/tutorials/search-beginners/) for science fiction books in just about 5 minutes with the help of Qdrant. You can also watch our [video tutorial](https://www.youtube.com/watch?v=AASiqmtKo54).
|
||||
Now that you're familiar with the core concepts around vector databases, it’s time to get your hands dirty. [Start by building your own semantic search engine](/documentation/tutorials/search-beginners/) for science fiction books in just about 5 minutes with the help of Qdrant. You can also watch our [video tutorial](https://www.youtube.com/watch?v=AASiqmtKo54).
|
||||
|
||||
Feeling ready to dive into a more complex project? Take the next step and get started building an actual [Neural Search Service with a complete API and a dataset](/documentation/tutorials/neural-search/).
|
||||
|
||||
|
||||
@@ -38,7 +38,7 @@ As your data grows, you’ll need efficient ways to identify the most relevant i
|
||||
|
||||
**Vector databases** store information as **vector embeddings**. This format supports efficient similarity searches to retrieve relevant data for your query. For example, Qdrant is specifically designed to perform fast, even in scenarios dealing with billions of vectors.
|
||||
|
||||
This article will focus on RAG systems and architecture. If you’re interested in learning more about vector search, we recommend the following articles: [What is a Vector Database?](https://qdrant.tech/articles/what-is-a-vector-database/) and [What are Vector Embeddings?](https://qdrant.tech/articles/what-are-embeddings/).
|
||||
This article will focus on RAG systems and architecture. If you’re interested in learning more about vector search, we recommend the following articles: [What is a Vector Database?](/articles/what-is-a-vector-database/) and [What are Vector Embeddings?](/articles/what-are-embeddings/).
|
||||
|
||||
|
||||
## RAG architecture
|
||||
@@ -64,7 +64,7 @@ As shown in the image above, here’s the process:
|
||||
* Start with a _loader_ that gathers _documents_ containing your data. These documents could be anything from articles and books to web pages and social media posts.
|
||||
* Next, a _splitter_ divides the documents into smaller chunks, typically sentences or paragraphs.
|
||||
* This is because RAG models work better with smaller pieces of text. In the diagram, these are _document snippets_.
|
||||
* Each text chunk is then fed into an _embedding machine_. This machine uses complex algorithms to convert the text into [vector embeddings](https://qdrant.tech/articles/what-are-embeddings/).
|
||||
* Each text chunk is then fed into an _embedding machine_. This machine uses complex algorithms to convert the text into [vector embeddings](/articles/what-are-embeddings/).
|
||||
|
||||
All the generated vector embeddings are stored in a knowledge base of indexed information. This supports efficient retrieval of similar pieces of information when needed.
|
||||
|
||||
@@ -94,14 +94,14 @@ The classic approach is **keyword search**, which scans documents for the exact
|
||||
|
||||
[TF-IDF](https://en.wikipedia.org/wiki/Tf%E2%80%93idf) (Term Frequency-Inverse Document Frequency) and [BM25](https://en.wikipedia.org/wiki/Okapi_BM25) are two classic related algorithms. They're simple and computationally efficient. However, they can struggle with synonyms and don't always capture semantic similarities.
|
||||
|
||||
If you’re interested in going deeper, refer to our article on [Sparse Vectors](https://qdrant.tech/articles/sparse-vectors/).
|
||||
If you’re interested in going deeper, refer to our article on [Sparse Vectors](/articles/sparse-vectors/).
|
||||
|
||||
|
||||
##### Dense vector embeddings
|
||||
|
||||
This approach uses large language models like [BERT](https://en.wikipedia.org/wiki/BERT_(language_model)) to encode the query and passages into dense vector embeddings. These models are compact numerical representations that capture semantic meaning. Vector databases like Qdrant store these embeddings, allowing retrieval based on **semantic similarity** rather than just keywords using distance metrics like cosine similarity.
|
||||
|
||||
This allows the retriever to match based on semantic understanding rather than just keywords. So if I ask about "compounds that cause BO," it can retrieve relevant info about "molecules that create body odor" even if those exact words weren't used. We explain more about it in our [What are Vector Embeddings](https://qdrant.tech/articles/what-are-embeddings/) article.
|
||||
This allows the retriever to match based on semantic understanding rather than just keywords. So if I ask about "compounds that cause BO," it can retrieve relevant info about "molecules that create body odor" even if those exact words weren't used. We explain more about it in our [What are Vector Embeddings](/articles/what-are-embeddings/) article.
|
||||
|
||||
|
||||
#### Hybrid search
|
||||
@@ -121,7 +121,7 @@ Some common hybrid approaches include:
|
||||
* Considering both semantic vector closeness and statistical keyword patterns/weights in a combined scoring model.
|
||||
* Having multiple stages were different techniques. One example: start with an initial keyword retrieval, followed by semantic re-ranking, then a final re-ranking using even more complex models.
|
||||
|
||||
When you combine the powers of different search methods in a complementary way, you can provide higher quality, more comprehensive results. Check out our article on [Hybrid Search](https://qdrant.tech/articles/hybrid-search/) if you’d like to learn more.
|
||||
When you combine the powers of different search methods in a complementary way, you can provide higher quality, more comprehensive results. Check out our article on [Hybrid Search](/articles/hybrid-search/) if you’d like to learn more.
|
||||
|
||||
|
||||
### The Generator
|
||||
|
||||
@@ -26,7 +26,5 @@ Here are the principles we followed while designing these benchmarks:
|
||||
|
||||
</details>
|
||||
|
||||
</br>
|
||||
|
||||
Some of our experiment design decisions are described in the [F.A.Q Section](/benchmarks/#benchmarks-faq).
|
||||
Reach out to us on our [Discord channel](https://qdrant.to/discord) if you want to discuss anything related Qdrant or these benchmarks.
|
||||
|
||||
@@ -3,9 +3,9 @@ draft: false
|
||||
id: 1
|
||||
title: Single node benchmarks
|
||||
description: |
|
||||
We benchmarked several vector databases using various configurations of them on different datasets to check how the results may vary. Those datasets may have different vector dimensionality but also vary in terms of the distance function being used. We also tried to capture the difference we can expect while using some different configuration parameters, for both the engine itself and the search operation separately. </br> </br> <b> Updated: January 2024 </b>
|
||||
We benchmarked several vector databases using various configurations of them on different datasets to check how the results may vary. Those datasets may have different vector dimensionality but also vary in terms of the distance function being used. We also tried to capture the difference we can expect while using some different configuration parameters, for both the engine itself and the search operation separately. </br> </br> <b> Updated: January/June 2024 </b>
|
||||
single_node_title: Single node benchmarks
|
||||
single_node_data: /benchmarks/results-1-100-thread.json
|
||||
single_node_data: /benchmarks/results-1-100-thread-2024-06-15.json
|
||||
preview_image: /benchmarks/benchmark-1.png
|
||||
date: 2022-08-23
|
||||
weight: 2
|
||||
@@ -21,7 +21,7 @@ Most of the engines have improved since [our last run](/benchmarks/single-node-s
|
||||
* `Elasticsearch` has become considerably fast for many cases but it's very slow in terms of indexing time. It can be 10x slower when storing 10M+ vectors of 96 dimensions! (32mins vs 5.5 hrs)
|
||||
* `Milvus` is the fastest when it comes to indexing time and maintains good precision. However, it's not on-par with others when it comes to RPS or latency when you have higher dimension embeddings or more number of vectors.
|
||||
* `Redis` is able to achieve good RPS but mostly for lower precision. It also achieved low latency with single thread, however its latency goes up quickly with more parallel requests. Part of this speed gain comes from their custom protocol.
|
||||
* `Weaviate` has improved the least since our last run. Because of relative improvements in other engines, it has become one of the slowest in terms of RPS as well as latency.
|
||||
* `Weaviate` has improved the least since our last run.
|
||||
|
||||
## How to read the results
|
||||
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
---
|
||||
title: Qdrant Blog
|
||||
subtitle: Check out our latest posts
|
||||
description: A place to learn how to become an expert traveler through vector space. Subscribe and we will update you on features and news.
|
||||
email_placeholder: Enter your email
|
||||
subscribe_button: Subscribe
|
||||
features_title: Features and News
|
||||
search_placeholder: What are you Looking for?
|
||||
aliases: # There is no need to add aliases for future new tags and categories!
|
||||
- /tags
|
||||
- /tags/case-study
|
||||
@@ -67,4 +72,4 @@ aliases: # There is no need to add aliases for future new tags and categories!
|
||||
- /categories/vector-search
|
||||
- /categories/webinar
|
||||
- /categories/vector-space-talk
|
||||
---
|
||||
---
|
||||
|
||||
@@ -97,11 +97,11 @@ This technology integrates Kubernetes clusters from any setting - cloud, on-prem
|
||||
|
||||
<p align="center"><iframe width="560" height="315" src="https://www.youtube.com/embed/BF02jULGCfo" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" allowfullscreen></iframe></p>
|
||||
|
||||
[Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) marks a significant advancement in vector databases, offering the most flexible way to implement vector search.
|
||||
[Qdrant Hybrid Cloud](/hybrid-cloud/) marks a significant advancement in vector databases, offering the most flexible way to implement vector search.
|
||||
|
||||
You can test out Qdrant Hybrid Cloud today. Sign up or log into your [Qdrant Cloud account](https://cloud.qdrant.io/login) and get started in the **Hybrid Cloud** section.
|
||||
|
||||
Also, to learn more about Qdrant Hybrid Cloud read our [Official Release Blog](https://qdrant.tech/blog/hybrid-cloud/) or our [Qdrant Hybrid Cloud website](https://hybrid-cloud.qdrant.tech/). For additional technical insights, please read our [documentation](https://qdrant.tech/documentation/hybrid-cloud/).
|
||||
Also, to learn more about Qdrant Hybrid Cloud read our [Official Release Blog](/blog/hybrid-cloud/) or our [Qdrant Hybrid Cloud website](/hybrid-cloud/). For additional technical insights, please read our [documentation](/documentation/hybrid-cloud/).
|
||||
|
||||
#### Try it out!
|
||||
|
||||
|
||||
@@ -41,8 +41,8 @@ Ready to experience the benefits of Qdrant on Azure Marketplace? Getting started
|
||||
|
||||
1. **Visit the Azure Marketplace**: Navigate to [Qdrant's Marketplace listing](https://azuremarketplace.microsoft.com/en-en/marketplace/apps/qdrantsolutionsgmbh1698769709989.qdrant-db).
|
||||
2. **Deploy Qdrant**: Follow the simple deployment instructions to set up your instance.
|
||||
3. **Start Using Qdrant**: Once deployed, start exploring the [features and capabilities of Qdrant](https://qdrant.tech/documentation/concepts/) on Azure.
|
||||
4. **Read Documentation**: Read Qdrant's [Documentation](https://qdrant.tech/documentation/) and build demo apps using [Tutorials](https://qdrant.tech/documentation/tutorials/).
|
||||
3. **Start Using Qdrant**: Once deployed, start exploring the [features and capabilities of Qdrant](/documentation/concepts/) on Azure.
|
||||
4. **Read Documentation**: Read Qdrant's [Documentation](/documentation/) and build demo apps using [Tutorials](/documentation/tutorials/).
|
||||
|
||||
## Join Us on this Exciting Journey:
|
||||
|
||||
|
||||
@@ -1,11 +1,10 @@
|
||||
---
|
||||
draft: false
|
||||
title: Batch vector search with Qdrant
|
||||
title: Mastering Batch Search for Vector Optimization | Qdrant
|
||||
slug: batch-vector-search-with-qdrant
|
||||
short_description: Introducing efficient batch vector search capabilities,
|
||||
streamlining and optimizing large-scale searches for enhanced performance.
|
||||
description: "Discover the latest feature designed to streamline and optimize
|
||||
large-scale searches. "
|
||||
description: "Discover how to optimize your vector search capabilities with efficient batch search. Learn optimization strategies for faster, more accurate results."
|
||||
preview_image: /blog/from_cms/andrey.vasnetsov_career_mining_on_the_moon_with_giant_machines_813bc56a-5767-4397-9243-217bea869820.png
|
||||
date: 2022-09-26T15:39:53.751Z
|
||||
author: Kacper Łukawski
|
||||
@@ -16,17 +15,20 @@ tags:
|
||||
- Machine Learning
|
||||
- Information Retrieval
|
||||
---
|
||||
The latest release of Qdrant 0.10.0 has introduced a lot of functionalities that simplify some common tasks. Those new possibilities come with some slightly modified interfaces of the client library. One of the recently introduced features is the possibility to query the collection with multiple vectors at once — a batch search mechanism.
|
||||
|
||||
# How to Optimize Vector Search Using Batch Search in Qdrant 0.10.0
|
||||
|
||||
The latest release of Qdrant 0.10.0 has introduced a lot of functionalities that simplify some common tasks. Those new possibilities come with some slightly modified interfaces of the client library. One of the recently introduced features is the possibility to query the collection with [multiple vectors](https://qdrant.tech/blog/storing-multiple-vectors-per-object-in-qdrant/) at once — a batch search mechanism.
|
||||
|
||||
There are a lot of scenarios in which you may need to perform multiple non-related tasks at the same time. Previously, you only could send several requests to Qdrant API on your own. But multiple parallel requests may cause significant network overhead and slow down the process, especially in case of poor connection speed.
|
||||
|
||||
Now, thanks to the new batch search, you don’t need to worry about that. Qdrant will handle multiple search requests in just one API call and will perform those requests in the most optimal way.
|
||||
|
||||
## An example of using the batch search
|
||||
## An example of using batch search to optimize vector search
|
||||
|
||||
We’ve used the official Python client to show how the batch search might be integrated with your application. Since there have been some changes in the interfaces of Qdrant 0.10.0, we’ll go step by step.
|
||||
|
||||
## Creating the collection
|
||||
### Step 1: Creating the collection
|
||||
|
||||
The first step is to create a collection with a specified configuration — at least vector size and the distance function used to measure the similarity between vectors.
|
||||
|
||||
@@ -41,7 +43,7 @@ client.recreate_collection(
|
||||
)
|
||||
```
|
||||
|
||||
## Loading the vectors
|
||||
## Step 2: Loading the vectors
|
||||
|
||||
With the collection created, we can put some vectors into it. We’re going to have just a few examples.
|
||||
|
||||
@@ -64,7 +66,7 @@ client.upload_collection(
|
||||
)
|
||||
```
|
||||
|
||||
## Batch search in a single request
|
||||
## Step 3: Batch search in a single request
|
||||
|
||||
Now we’re ready to start looking for similar vectors, as our collection has some entries. Let’s say we want to find the distance between the selected vector and the most similar database entry and at the same time find the two most similar objects for a different vector query. Up till 0.9, we would need to call the API twice. Now, we can send both requests together:
|
||||
|
||||
@@ -99,7 +101,7 @@ Each instance of the SearchRequest class may provide its own search parameters,
|
||||
|
||||
And that’s it! You no longer have to handle the multiple requests on your own. Qdrant will do it under the hood.
|
||||
|
||||
## Benchmark
|
||||
## Batch Search Benchmarks
|
||||
|
||||
The batch search is fairly easy to be integrated into your application, but if you prefer to see some numbers before deciding to switch, then it’s worth comparing four different options:
|
||||
|
||||
@@ -125,4 +127,6 @@ Additional improvements could be achieved in the case of distributed deployment,
|
||||
|
||||
## Summary
|
||||
|
||||
Batch search allows packing different queries into a single API call and retrieving the results in a single response. If you ever struggled with sending several consecutive queries into Qdrant, then you can easily switch to the new batch search method and simplify your application code. As shown in the benchmarks, that may almost effortlessly speed up your interactions with Qdrant even by over 30%, even not considering the spare network overhead and possible reuse of filters!
|
||||
Batch search allows packing different queries into a single API call and retrieving the results in a single response. If you ever struggled with sending several consecutive queries into Qdrant, then you can easily switch to the new batch search method and simplify your application code. As shown in the benchmarks, that may almost effortlessly speed up your interactions with Qdrant even by over 30%, even not considering the spare network overhead and possible reuse of filters!
|
||||
|
||||
Ready to unlock the potential of batch search and optimize your vector search with Qdrant 0.10.0? Contact us today to learn how we can revolutionize your search capabilities!
|
||||
@@ -38,7 +38,7 @@ more time shipping features and fixing bugs.
|
||||
bloop’s mission is to make software engineers autonomous and semantic code search is the cornerstone
|
||||
of that vision. The project is maintained by a group of Rust and Typescript engineers and ML researchers.
|
||||
It leverages many prominent nascent technologies, such as [Tauri](http://tauri.app), [tantivy](https://docs.rs/tantivy),
|
||||
[Qdrant](https://qdrant.tech) and [Anthropic](https://www.anthropic.com/).
|
||||
[Qdrant](https://github.com/qdrant/qdrant) and [Anthropic](https://www.anthropic.com/).
|
||||
|
||||
## About Qdrant
|
||||
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Kairoswealth & Qdrant: Transforming Wealth Management with AI-Driven Insights and Scalable Vector Search"
|
||||
short_description: "Transforming wealth management with AI-driven insights and scalable vector search."
|
||||
description: "Enhancing wealth management using AI-driven insights and efficient vector search for improved recommendations and scalability."
|
||||
preview_image: /blog/case-study-kairoswealth/preview.png
|
||||
social_preview_image: /blog/case-study-kairoswealth/preview.png
|
||||
date: 2024-07-10T00:02:00Z
|
||||
author: Qdrant
|
||||
featured: false
|
||||
tags:
|
||||
- Kairoswealth
|
||||
- Vincent Teyssier
|
||||
- AI-Driven Insights
|
||||
- Performance Scalability
|
||||
- Multi-Tenancy
|
||||
- Financial Recommendations
|
||||
---
|
||||
|
||||

|
||||
|
||||
### **About Kairoswealth**
|
||||
|
||||
[Kairoswealth](https://kairoswealth.com/) is a comprehensive wealth management platform designed to provide users with a holistic view of their financial portfolio. The platform offers access to unique financial products and automates back-office operations through its AI assistant, Gaia.
|
||||
|
||||

|
||||
|
||||
### **Motivations for Adopting a Vector Database**
|
||||
|
||||
“At Kairoswealth we encountered several use cases necessitating the ability to run similarity queries on large datasets. Key applications included product recommendations and retrieval-augmented generation (RAG),” says [Vincent Teyssier](https://www.linkedin.com/in/vincent-teyssier/), Chief Technology & AI Officer at Kairoswealth. These needs drove the search for a more robust and scalable vector database solution.
|
||||
|
||||
### **Challenges with Previous Solutions**
|
||||
|
||||
“We faced several critical showstoppers with our previous vector database solution, which led us to seek an alternative,” says Teyssier. These challenges included:
|
||||
|
||||
- **Performance Scalability:** Significant performance degradation occurred as more data was added, despite various optimizations.
|
||||
- **Robust Multi-Tenancy:** The previous solution struggled with multi-tenancy, impacting performance.
|
||||
- **RAM Footprint:** High memory consumption was an issue.
|
||||
|
||||
### **Qdrant Use Cases at Kairoswealth**
|
||||
|
||||
Kairoswealth leverages Qdrant for several key use cases:
|
||||
|
||||
- **Internal Data RAG:** Efficiently handling internal RAG use cases.
|
||||
- **Financial Regulatory Reports RAG:** Managing and generating financial reports.
|
||||
- **Recommendations:** Enhancing the accuracy and efficiency of recommendations with the Kairoswealth platform.
|
||||
|
||||

|
||||
|
||||
### **Why Kairoswealth Chose Qdrant**
|
||||
|
||||
Some of the key reasons, why Kairoswealth landed on Qdrant as the vector database of choice are:
|
||||
|
||||
1. **High Performance with 2.4M Vectors:** “Qdrant efficiently handled the indexing of 1.2 million vectors with 16 metadata fields each, maintaining high performance with no degradation. Similarity queries and scrolls run in less than 0.3 seconds. When we doubled the dataset to 2.4 million vectors, performance remained consistent.So we decided to double that to 2.4M vectors, and it's as if we were inserting our first vector!” says Teyssier.
|
||||
2. **8x Memory Efficiency:** The database storage size with Qdrant was eight times smaller than the previous solution, enabling the deployment of the entire dataset on smaller instances and saving significant infrastructure costs.
|
||||
3. **Embedded Capabilities:** “Beyond simple search and similarity, Qdrant hosts a bunch of very nice features around recommendation engines, adding positive and negative examples for better spacial narrowing, efficient multi-tenancy, and many more,” says Teyssier.
|
||||
4. **Support and Community:** “The Qdrant team, led by Andre Zayarni, provides exceptional support and has a strong passion for data engineering,” notes Teyssier, “the team's commitment to open-source and their active engagement in helping users, from beginners to veterans, is highly valued by Kairoswealth.”
|
||||
|
||||
### **Conclusion**
|
||||
|
||||
Kairoswealth's transition to Qdrant has enabled them to overcome significant challenges related to performance, scalability, and memory efficiency, while also benefiting from advanced features and robust support. This partnership positions Kairoswealth to continue innovating in the wealth management sector, leveraging the power of AI to deliver superior services to their clients.
|
||||
|
||||
### **Future Roadmap for Kairoswealth**
|
||||
|
||||
Kairoswealth is seizing the opportunity to disrupt the wealth management sector, which has traditionally been underserved by technology. For example, they are developing the Kairos Terminal, a natural language interface that translates user queries into OpenBB commands (a set of tools for financial analysis and data visualization within the OpenBB Terminal). With regards to the future of the wealth management sector, Teyssier notes that “the integration of Generative AI will automate back-office tasks such as data collation, data reconciliation, and market research. This technology will also enable wealth managers to scale their services to broader segments, including affluent clients, by automating relationship management and interactions.”
|
||||
@@ -0,0 +1,80 @@
|
||||
---
|
||||
title: "Community Highlights #1"
|
||||
draft: false
|
||||
slug: community-highlights-1 # Change this slug to your page slug if needed
|
||||
short_description: Celebrating top contributions and achievements in vector search, featuring standout projects, articles, and the Creator of the Month, Pavan Kumar. # Change this
|
||||
description: Celebrating top contributions and achievements in vector search, featuring standout projects, articles, and the Creator of the Month, Pavan Kumar!
|
||||
preview_image: /blog/community-highlights-1/preview-image.png
|
||||
social_preview_image: /blog/community-highlights-1/preview-image.png
|
||||
|
||||
date: 2024-06-20T11:57:37-03:00
|
||||
author: Sabrina Aquino
|
||||
featured: false
|
||||
tags:
|
||||
- news
|
||||
- vector search
|
||||
- qdrant
|
||||
- ambassador program
|
||||
- community
|
||||
- artificial intelligence
|
||||
---
|
||||
|
||||
Welcome to the very first edition of Community Highlights, where we celebrate the most impactful contributions and achievements of our vector search community! 🎉
|
||||
|
||||
## Content Highlights 🚀
|
||||
|
||||
Here are some standout projects and articles from our community this past month. If you're looking to learn more about vector search or build some great projects, we recommend you to check these guides:
|
||||
|
||||
* **[Implementing Advanced Agentic Vector Search](https://towardsdev.com/implementing-advanced-agentic-vector-search-a-comprehensive-guide-to-crewai-and-qdrant-ca214ca4d039): A Comprehensive Guide to CrewAI and Qdrant by [Pavan Kumar](https://www.linkedin.com/in/kameshwara-pavan-kumar-mantha-91678b21/)**
|
||||
* **Build Your Own RAG Using [Unstructured, Llama3 via Groq, Qdrant & LangChain](https://www.youtube.com/watch?v=m_3q3XnLlTI) by [Sudarshan Koirala](https://www.linkedin.com/in/sudarshan-koirala/)**
|
||||
* **Qdrant filtering and [self-querying retriever](https://www.youtube.com/watch?v=iaXFggqqGD0) retrieval with LangChain by [Daniel Romero](https://www.linkedin.com/in/infoslack/)**
|
||||
* **RAG Evaluation with [Arize Phoenix](https://superlinked.com/vectorhub/articles/retrieval-augmented-generation-eval-qdrant-arize) by [Atita Arora](https://www.linkedin.com/in/atitaarora/)**
|
||||
* **Building a Serverless Application with [AWS Lambda and Qdrant](https://medium.com/@benitomartin/building-a-serverless-application-with-aws-lambda-and-qdrant-for-semantic-search-ddb7646d4c2f) for Semantic Search by [Benito Martin](https://www.linkedin.com/in/benitomzh/)**
|
||||
* **Production ready Secure and [Powerful AI Implementations with Azure Services](https://towardsdev.com/production-ready-secure-and-powerful-ai-implementations-with-azure-services-671b68631212) by [Pavan Kumar](https://www.linkedin.com/in/kameshwara-pavan-kumar-mantha-91678b21/)**
|
||||
* **Building [Agentic RAG with Rust, OpenAI & Qdrant](https://medium.com/@joshmo_dev/building-agentic-rag-with-rust-openai-qdrant-d3a0bb85a267) by [Joshua Mo](https://www.linkedin.com/in/joshua-mo-4146aa220/)**
|
||||
* **Qdrant [Hybrid Search](https://medium.com/@nickprock/qdrant-hybrid-search-under-the-hood-using-haystack-355841225ac6) under the hood using Haystack by [Nicola Procopio](https://www.linkedin.com/in/nicolaprocopio/)**
|
||||
* **[Llama 3 Powered Voice Assistant](https://medium.com/@datadrifters/llama-3-powered-voice-assistant-integrating-local-rag-with-qdrant-whisper-and-langchain-b4d075b00ac5): Integrating Local RAG with Qdrant, Whisper, and LangChain by [Datadrifters](https://medium.com/@datadrifters)**
|
||||
* **[Distributed deployment](https://medium.com/@vardhanam.daga/distributed-deployment-of-qdrant-cluster-with-sharding-replicas-e7923d483ebc) of Qdrant cluster with sharding & replicas by [Vardhanam Daga](https://www.linkedin.com/in/vardhanam-daga/overlay/about-this-profile/)**
|
||||
* **Private [Healthcare AI Assistant](https://medium.com/aimpact-all-things-ai/building-private-healthcare-ai-assistant-for-clinics-using-qdrant-hybrid-cloud-jwt-rbac-dspy-and-089a772e08ae) using Qdrant Hybrid Cloud, DSPy, and Groq by [Sachin Khandewal](https://www.linkedin.com/in/sachink1729/)**
|
||||
|
||||
|
||||
## Creator of the Month 🌟
|
||||
|
||||
|
||||
<img src="/blog/community-highlights-1/creator-of-the-month-pavan.png" alt="Picture of Pavan Kumar with over 6 content contributions for the Creator of the Month" style="width: 70%;" />
|
||||
|
||||
|
||||
Congratulations to Pavan Kumar for being awarded **Creator of the Month!** Check out what were Pavan's most valuable contributions to the Qdrant vector search community this past month:
|
||||
|
||||
|
||||
* **[Implementing Advanced Agentic Vector Search](https://towardsdev.com/implementing-advanced-agentic-vector-search-a-comprehensive-guide-to-crewai-and-qdrant-ca214ca4d039): A Comprehensive Guide to CrewAI and Qdrant**
|
||||
* **Production ready Secure and [Powerful AI Implementations with Azure Services](https://towardsdev.com/production-ready-secure-and-powerful-ai-implementations-with-azure-services-671b68631212)**
|
||||
* **Building Neural Search Pipelines with Azure and Qdrant: A Step-by-Step Guide [Part-1](https://towardsdev.com/building-neural-search-pipelines-with-azure-and-qdrant-a-step-by-step-guide-part-1-40c191084258) and [Part-2](https://towardsdev.com/building-neural-search-pipelines-with-azure-and-qdrant-a-step-by-step-guide-part-2-fba287b49574)**
|
||||
* **Building a RAG System with [Ollama, Qdrant and Raspberry Pi](https://blog.gopenai.com/harnessing-ai-at-the-edge-building-a-rag-system-with-ollama-qdrant-and-raspberry-pi-45ac3212cf75)**
|
||||
* **Building a [Multi-Document ReAct Agent](https://blog.stackademic.com/building-a-multi-document-react-agent-for-financial-analysis-using-llamaindex-and-qdrant-72a535730ac3) for Financial Analysis using LlamaIndex and Qdrant**
|
||||
|
||||
Pavan is a seasoned technology expert with 14 years of extensive experience, passionate about sharing his knowledge through technical blogging, engaging in technical meetups, and staying active with cycling!
|
||||
|
||||
Thank you, Pavan, for your outstanding contributions and commitment to the community!
|
||||
|
||||
## Most Active Members 🏆
|
||||
|
||||
|
||||
<img src="/blog/community-highlights-1/most-active-members.png" alt="Picture of the 3 most active members of our vector search community" style="width: 70%;" />
|
||||
|
||||
|
||||
We're excited to recognize our most active community members, who have been a constant support to vector search builders, and sharing their knowledge and making our community more engaging:
|
||||
|
||||
* 🥇 **1st Place: Robert Caulk**
|
||||
* 🥈 **2nd Place: Nicola Procopio**
|
||||
* 🥉 **3rd Place: Joshua Mo**
|
||||
|
||||
Thank you all for your dedication and for making the Qdrant vector search community such a dynamic and valuable place!
|
||||
|
||||
Stay tuned for more highlights and updates in the next edition of Community Highlights! 🚀
|
||||
|
||||
**Join us for Office Hours! 🎙️**
|
||||
|
||||
Don't miss our next [Office Hours hangout on Discord](https://discord.gg/s9YxGeQK?event=1252726857753821236), happening next week on June 27th. This is a great opportunity to introduce yourself to the community, learn more about vector search, and engage with the people behind this awesome content!
|
||||
|
||||
See you there 👋
|
||||
@@ -0,0 +1,176 @@
|
||||
---
|
||||
title: "Comparing Qdrant vs Pinecone: A Detailed Analysis of Vector Databases for AI Applications"
|
||||
draft: false
|
||||
short_description: "Highlighting performance, features, and suitability for various use cases."
|
||||
description: "This comprehensive comparison highlights performance, features, and suitability for various use cases."
|
||||
preview_image: /blog/comparing-qdrant-vs-pinecone-vector-databases/social_preview.png
|
||||
social_preview_image: /blog/comparing-qdrant-vs-pinecone-vector-databases/social_preview.png
|
||||
aliases: /documentation/overview/qdrant-alternatives/
|
||||
date: 2024-02-25T00:00:00-08:00
|
||||
author: Qdrant Team
|
||||
featured: false
|
||||
tags:
|
||||
- vector search
|
||||
- role based access control
|
||||
- byte vectors
|
||||
- binary vectors
|
||||
- quantization
|
||||
- new features
|
||||
---
|
||||
|
||||
# Comparing Qdrant vs Pinecone: Vector Database Showdown
|
||||
|
||||
Data forms the foundation upon which AI applications are built. Data can exist in both structured and unstructured formats. Structured data typically has well-defined schemas or inherent relationships. However, unstructured data, such as text, image, audio, or video, must first be converted into numerical representations known as [vector embeddings](https://qdrant.tech/articles/what-are-embeddings/). These embeddings encapsulate the semantic meaning or features of unstructured data and are in the form of high-dimensional vectors.
|
||||
|
||||
Traditional databases, while effective at handling structured data, fall short when dealing with high-dimensional unstructured data, which are increasingly the focal point of modern AI applications. Key reasons include:
|
||||
|
||||
- **Indexing Limitations**: Database indexing methods like B-Trees or hash indexes, typically used in relational databases, are inefficient for high-dimensional data and show poor query performance.
|
||||
- **Curse of Dimensionality**: As dimensions increase, data points become sparse, and distance metrics like Euclidean distance lose their effectiveness, leading to poor search query performance.
|
||||
- **Lack of Specialized Algorithms**: Traditional databases do not incorporate advanced algorithms designed to handle high-dimensional data, resulting in slow query processing times.
|
||||
- **Scalability Challenges**: Managing and querying high-dimensional vectors require optimized data structures, which traditional databases are not built to handle.
|
||||
- **Storage Inefficiency**: Traditional databases are not optimized for efficiently storing large volumes of high-dimensional data, facing significant challenges in managing space complexity and retrieval efficiency.
|
||||
|
||||
Vector databases address these challenges by efficiently storing and querying high-dimensional vectors. They offer features such as high-dimensional vector storage and retrieval, efficient similarity search, sophisticated indexing algorithms, advanced compression techniques, and integration with various machine learning frameworks.
|
||||
|
||||
Due to their capabilities, vector databases are now a cornerstone of modern AI and are becoming pivotal in building applications that leverage similarity search, recommendation systems, natural language processing, computer vision, image recognition, speech recognition, and more.
|
||||
|
||||
Over the past few years, several vector database solutions have emerged – the two leading ones being Qdrant and Pinecone, among others. Both are powerful vector database solutions with unique strengths. However, they differ greatly in their principles and approach, and the capabilities they offer to developers. In this article, we’ll examine both solutions and discuss the factors you need to consider when choosing amongst the two. Let’s dive in!
|
||||
|
||||
## Exploring Qdrant Vector Database: Features and Capabilities
|
||||
|
||||
Qdrant is a high-performance, open-source vector similarity search engine built with Rust, designed to handle the demands of large-scale AI applications with exceptional speed and reliability. Founded in 2021, Qdrant's mission is to "build the most efficient, scalable, and high-performance vector database in the market." This mission is reflected in its architecture and feature set.
|
||||
|
||||
Qdrant is highly scalable and performant: it can handle billions of vectors efficiently and with [minimal latency](https://qdrant.tech/benchmarks/). Its advanced vector indexing, search, and retrieval capabilities make it ideal for applications that require fast and accurate search results. It supports vertical and horizontal scaling, advanced compression techniques, highly flexible deployment options – including cloud-native, hybrid cloud, and private cloud solutions – and powerful security features.
|
||||
|
||||
Let’s look at some of its key features.
|
||||
|
||||
- **Advanced Similarity Search:** Qdrant supports various similarity [search](https://qdrant.tech/documentation/concepts/search/) metrics like dot product, cosine similarity, Euclidean distance, and Manhattan distance. You can store additional information along with vectors, known as [payload](https://qdrant.tech/documentation/concepts/payload/) in Qdrant terminology. A payload is any JSON formatted data.
|
||||
- **Built Using Rust:** Qdrant is built with Rust, and leverages its performance and efficiency. Rust is famed for its [memory safety](https://arxiv.org/abs/2206.05503) without the overhead of a garbage collector, and rivals C and C++ in speed.
|
||||
- **Scaling and Multitenancy**: Qdrant supports both vertical and horizontal scaling and uses the Raft consensus protocol for [distributed deployments](https://qdrant.tech/documentation/guides/distributed_deployment/). Developers can run Qdrant clusters with replicas and shards, and seamlessly scale to handle large datasets. Qdrant also supports [multitenancy](https://qdrant.tech/documentation/guides/multiple-partitions/) where developers can create single collections and partition them using payload.
|
||||
- **Payload Indexing and Filtering:** Just as Qdrant allows attaching any JSON payload to vectors, it also supports payload indexing and [filtering](https://qdrant.tech/documentation/concepts/filtering/) with a wide range of data types and query conditions, including keyword matching, full-text filtering, numerical ranges, nested object filters, and [geo](https://qdrant.tech/documentation/concepts/filtering/#geo)filtering.
|
||||
- **Hybrid Search with Sparse Vectors:** Qdrant supports both dense and [sparse vectors](https://qdrant.tech/articles/sparse-vectors/), thereby enabling hybrid search capabilities. Sparse vectors are numerical representations of data where most of the elements are zero. Developers can combine search results from dense and sparse vectors, where sparse vectors ensure that results containing the specific keywords are returned and dense vectors identify semantically similar results.
|
||||
- **Built-In Vector Quantization:** Qdrant offers three different [quantization](https://qdrant.tech/documentation/guides/quantization/) options to developers to optimize resource usage. Scalar quantization balances accuracy, speed, and compression by converting 32-bit floats to 8-bit integers. Binary quantization, the fastest method, significantly reduces memory usage. Product quantization offers the highest compression, and is perfect for memory-constrained scenarios.
|
||||
- **Flexible Deployment Options:** Qdrant offers a range of deployment options. Developers can easily set up Qdrant (or Qdrant cluster) [locally](https://qdrant.tech/documentation/quick-start/#download-and-run) using Docker for free. [Qdrant Cloud](https://qdrant.tech/cloud/), on the other hand, is a scalable, managed solution that provides easy access with flexible pricing. Additionally, Qdrant offers [Hybrid Cloud](https://qdrant.tech/hybrid-cloud/) which integrates Kubernetes clusters from cloud, on-premises, or edge, into an enterprise-grade managed service.
|
||||
- **Security through API Keys, JWT and RBAC:** Qdrant offers developers various ways to [secure](https://qdrant.tech/documentation/guides/security/) their instances. For simple authentication, developers can use API keys (including Read Only API keys). For more granular access control, it offers JSON Web Tokens (JWT) and the ability to build Role-Based Access Control (RBAC). TLS can be enabled to secure connections. Qdrant is also [SOC 2 Type II](https://qdrant.tech/blog/qdrant-soc2-type2-audit/) certified.
|
||||
|
||||
Additionally, Qdrant integrates seamlessly with popular machine learning frameworks such as LangChain, LlamaIndex, and Haystack; and Qdrant Hybrid Cloud integrates seamlessly with AWS, DigitalOcean, Google Cloud, Linode, Oracle Cloud, OpenShift, and Azure, among others.
|
||||
|
||||
By focusing on performance, scalability and efficiency, Qdrant has positioned itself as a leading solution for enterprise-grade vector similarity search, capable of meeting the growing demands of modern AI applications.
|
||||
|
||||
However, how does it compare with Pinecone? Let’s take a look.
|
||||
|
||||
## Exploring Pinecone Vector Database: Key Features and Capabilities
|
||||
|
||||
Pinecone provides a fully managed vector database that abstracts the complexities of infrastructure and scaling. The company’s founding principle, when it started in 2019, was to make Pinecone “accessible to engineering teams of all sizes and levels of AI expertise.”
|
||||
|
||||
Similarly to Qdrant, Pinecone offers advanced vector search and retrieval capabilities. There are two different ways you can use Pinecone: using its serverless architecture or its pod architecture. Pinecone also supports advanced similarity search metrics such as dot product, Euclidean distance, and cosine similarity. Using its pod architecture, you can leverage horizontal or vertical scaling. Finally, Pinecone offers privacy and security features such as Role-Based Access Control (RBAC) and end-to-end encryption, including encryption in transit and at rest.
|
||||
|
||||
Let’s take a closer look at Pinecone’s features.
|
||||
|
||||
- **Fully Managed Service:** Pinecone offers a fully managed SaaS-only service. It handles the complexities of infrastructure management such as scaling, performance optimization, and maintenance. Pinecone is designed for developers who want to focus on building AI applications without worrying about the underlying database infrastructure.
|
||||
- **Serverless and Pod Architecture:** Pinecone offers two different architecture options to run their vector database - the serverless architecture and the pod architecture. Serverless architecture runs as a managed service on the AWS cloud platform, and allows automatic scaling based on workload. Pod architecture, on the other hand, provides pre-configured hardware units (pods) for hosting and executing services, and supports horizontal and vertical scaling. Pods can be run on AWS, GCP, or Azure.
|
||||
- **Advanced Similarity Search:** Pinecone supports three different similarity search metrics – dot product, Euclidean distance, and cosine similarity. It currently does not support Manhattan distance metric.
|
||||
- **Privacy and Security Features:** Pinecone offers Role-Based Access Control (RBAC), end-to-end encryption, and compliance with SOC 2 Type II and GDPR. Pinecone allows for the creation of “organization”, which, in turn, has “projects” and “members” with single sign-on (SSO) and access control.
|
||||
- **Hybrid Search and Sparse Vectors**: Pinecone supports both sparse and dense vectors, and allows hybrid search. This gives developers the ability to combine semantic and keyword search in a single query.
|
||||
- **Metadata Filtering**: Pinecone allows attaching key-value metadata to vectors in an index, which can later be queried. Semantic search using metadata filters retrieve exactly the results that match the filters.
|
||||
|
||||
Pinecone’s fully managed service makes it a compelling choice for developers who’re looking for a vector database that comes without the headache of infrastructure management.
|
||||
|
||||
## Pinecone vs Qdrant: Key Differences and Use Cases
|
||||
|
||||
Qdrant and Pinecone are both robust vector database solutions, but they differ significantly in their design philosophy, deployment options, and technical capabilities.
|
||||
|
||||
Qdrant is an open-source vector database that gives control to the developer. It can be run locally, on-prem, in the cloud, or as a managed service, and it even offers a hybrid cloud option for enterprises. This makes Qdrant suitable for a wide range of environments, from development to enterprise settings. It supports multiple programming languages and offers advanced features like customizable distance metrics, payload filtering, and integration with popular AI frameworks.
|
||||
|
||||
Pinecone, on the other hand, is a fully managed, SaaS-only solution designed to abstract the complexities of infrastructure management. It provides a serverless architecture for automatic scaling and a pod architecture for resource customization. Pinecone focuses on ease of use and high performance, offering built-in security measures, compliance certifications, and a user-friendly API. However, it has some limitations in terms of metadata handling and flexibility compared to Qdrant.
|
||||
|
||||
| Aspect | Qdrant | Pinecone |
|
||||
| ------------------------- | ---------------------------------------------------------------------- | -------------------------------------------------- |
|
||||
| Deployment Modes | Local, on-premises, cloud | SaaS-only |
|
||||
| Supported Languages | Python, JavaScript/TypeScript, Rust, Go, Java | Python, JavaScript/TypeScript, Java, Go |
|
||||
| Similarity Search Metrics | Dot Product, Cosine Similarity, Euclidean Distance, Manhattan Distance | Dot Product, Cosine Similarity, Euclidean Distance |
|
||||
|
||||
| Hybrid
|
||||
Search | Highly customizable Hybrid search by combining Sparse and Dense Vectors, with support for separate indices within the same collection | Supports Hybrid search with a single sparse-dense index |
|
||||
| Vector Payload | Accepts any JSON object as payload, supports NULL values, geolocation, and multiple vectors per point | Flat metadata structure, does not support NULL values, geolocation, or multiple vectors per point |
|
||||
| Scalability | Vertical and horizontal scaling, distributed deployment with Raft consensus | Serverless architecture and pod architecture for horizontal and vertical scaling |
|
||||
| Performance | Efficient indexing, low latency, high throughput, customizable distance metrics | High throughput, low latency, gRPC client for higher upsert speeds |
|
||||
| Security | Flexible, environment-specific configurations, API key authentication in Qdrant Cloud, JWT and RBAC, SOC 2 Type II certification | Built-in RBAC, end-to-end encryption, SOC 2 Type II certification |
|
||||
|
||||
## Making the Right Choice: Factors to Consider
|
||||
|
||||
When choosing between Qdrant and Pinecone, you need to consider some key factors that may impact your project long-term. Below are some primary considerations to help guide your decision:
|
||||
|
||||
### **1. Deployment Flexibility**
|
||||
|
||||
**Qdrant** offers multiple deployment options, including a local Docker node or cluster, Qdrant Cloud, and Hybrid Cloud. This allows you to choose an environment that best suits your project. You can start with a local Docker node for development, then add nodes to your cluster, and later switch to a Hybrid Cloud solution.
|
||||
|
||||
**Pinecone**, on the other hand, is a fully managed SaaS solution. To use Pinecone, you connect your development environment to its cloud service. It abstracts the complexities of infrastructure management, making it easier to deploy, but it is also less flexible in terms of deployment options compared to Qdrant.
|
||||
|
||||
### **2. Scalability Requirements**
|
||||
|
||||
**Qdrant** supports both vertical and horizontal scaling and is suitable for deployments of all scales. You can run it as a single Docker node, a large cluster, or a Hybrid cloud, depending on the size of your dataset. Qdrant’s architecture allows for distributed deployment with replicas and shards, and scales extremely well to billions of vectors with minimal latency.
|
||||
|
||||
**Pinecone** provides a serverless architecture and a pod architecture that automatically scales based on workload. Serverless architecture removes the need for any manual intervention, whereas pod architecture provides a bit more control. Since Pinecone is a managed SaaS-only solution, your application’s scalability is tied to both Pinecone's service and the underlying cloud provider in use.
|
||||
|
||||
### **3. Performance and Throughput**
|
||||
|
||||
**Qdrant** excels in providing different performance profiles tailored to specific use cases. It offers efficient vector and payload indexing, low-latency queries, optimizers, and high throughput, along with multiple options for quantization to further optimize performance.
|
||||
|
||||
**Pinecone** recommends increasing the number of replicas to boost the throughput of pod-based indexes. For serverless indexes, Pinecone automatically handles scaling and throughput. To decrease latency, Pinecone suggests using namespaces to partition records within a single index. However, since Pinecone is a managed SaaS-only solution, developer control over performance and throughput is limited.
|
||||
|
||||
### **4. Security Considerations**
|
||||
|
||||
**Qdrant** allows for tailored security configurations specific to your deployment environment. It supports API keys (including read-only API keys), JWT authentication, and TLS encryption for connections. Developers can build Role-Based Access Control (RBAC) according to their application needs in a completely custom manner. Additionally, Qdrant's deployment flexibility allows organizations that need to adhere to stringent data laws to deploy it within their infrastructure, ensuring compliance with data sovereignty regulations.
|
||||
|
||||
**Pinecone** provides comprehensive built-in security features in its managed SaaS solution, including Role-Based Access Control (RBAC) and end-to-end encryption. Its compliance with SOC 2 Type II and GDPR-readiness makes it a good choice for applications requiring standardized security measures.
|
||||
|
||||
### **5. Cost**
|
||||
|
||||
**Qdrant** can be self-hosted locally (single node or a cluster) with a single Docker command. With its SaaS option, it offers a free tier in Qdrant Cloud sufficient for around 1M 768-dimensional vectors, without any limitation on the number of collections it is used for. This allows developers to build multiple demos without limitations. For more pricing information, check [here](https://qdrant.tech/pricing/).
|
||||
|
||||
**Pinecone** cannot be self-hosted, and signing up for the SaaS solution is the only option. Pinecone has a free tier that supports approximately 300K 1536-dimensional embeddings. For Pinecone’s pricing details, check their pricing page.
|
||||
|
||||
### **Vector Database Comparison: A Summary**
|
||||
|
||||
The choice between Qdrant and Pinecone hinges on your specific needs:
|
||||
|
||||
- **Qdrant** is ideal for organizations that require flexible deployment options, extensive scalability, and customization. It is also suitable for projects needing deep integration with existing security infrastructure and those looking for a cost-effective, self-hosted solution.
|
||||
- **Pinecone** is suitable for teams seeking a fully managed solution with robust built-in security features and standardized compliance. It is suitable for cloud-native applications and dynamic environments where automatic scaling and low operational overhead are critical.
|
||||
|
||||
By carefully considering these factors, you can select the vector database that best aligns with your technical requirements and strategic goals.
|
||||
|
||||
## Choosing the Best Vector Database for Your AI Project
|
||||
|
||||
Selecting the best vector database for your AI project depends on several factors, including your deployment preferences, scalability needs, performance requirements, and security considerations.
|
||||
|
||||
- **Choose Qdrant if**:
|
||||
- You require flexible deployment options (local, on-premises, managed SaaS solution, or a Hybrid Cloud).
|
||||
- You need extensive customization and control over your vector database.
|
||||
- You project needs to adhere to data security and data sovereignty laws specific to your geography
|
||||
- Your project would benefit from advanced search capabilities, including complex payload filtering and geolocation support.
|
||||
- Cost efficiency and the ability to self-host are significant considerations.
|
||||
- **Choose Pinecone if**:
|
||||
- You prefer a fully managed SaaS solution that abstracts the complexities of infrastructure management.
|
||||
- You need a serverless architecture that automatically adjusts to varying workloads.
|
||||
- Built-in security features and compliance certifications (SOC 2 Type II, GDPR) are sufficient for your application.
|
||||
- You want to build your project with minimal operational overhead.
|
||||
|
||||
For maximum control, security, and cost-efficiency, choose Qdrant. It offers flexible deployment options, customizability, and advanced search features, and is ideal for building data sovereign AI applications. However, if you prioritize ease of use and automatic scaling with built-in security, Pinecone's fully managed SaaS solution with a serverless architecture is the way to go.
|
||||
|
||||
## Next Steps
|
||||
|
||||
Qdrant is one of the leading Pinecone alternatives in the market. For developers who seek control of their vector database, Qdrant offers the highest level of customization, flexible deployment options, and advanced security features.
|
||||
|
||||
To get started with Qdrant, explore our [documentation](https://qdrant.tech/documentation/), hop on to our [Discord](https://qdrant.to/discord) channel, sign up for [Qdrant cloud](https://cloud.qdrant.io/) (or [Hybrid cloud](https://qdrant.tech/hybrid-cloud/)), or [get in touch](https://qdrant.tech/contact-us/) with us today.
|
||||
|
||||
References:
|
||||
|
||||
- [Pinecone Documentation](https://docs.pinecone.io/)
|
||||
- [Qdrant Documentation](https://qdrant.tech/documentation/)
|
||||
|
||||
- If you aren't ready yet, [try out Qdrant locally](/documentation/quick-start/) or sign up for [Qdrant Cloud](https://cloud.qdrant.io/).
|
||||
|
||||
- For more basic information on Qdrant read our [Overview](/documentation/overview/) section or learn more about Qdrant Cloud's [Free Tier](/documentation/cloud/).
|
||||
|
||||
- If ready to migrate, please consult our [Comprehensive Guide](https://github.com/NirantK/qdrant_tools) for further details on migration steps.
|
||||
@@ -3,7 +3,7 @@ title: "Response to CVE-2024-2221: Arbitrary file upload vulnerability"
|
||||
draft: false
|
||||
slug: cve-2024-2221-response
|
||||
short_description: Qdrant keeps your systems secure
|
||||
description: Upgrade your deployments to at least v1.8.0. Cloud deployments not materially affected.
|
||||
description: Upgrade your deployments to at least v1.9.0. Cloud deployments not materially affected.
|
||||
preview_image: /blog/cve-2024-2221/cve-2024-2221-response-social-preview.png
|
||||
|
||||
# social_preview_image: /blog/Article-Image.png # Optional image used for link previews
|
||||
@@ -22,7 +22,7 @@ weight: 0 # Change this weight to change order of posts
|
||||
### Summary
|
||||
|
||||
A security vulnerability has been discovered in Qdrant affecting all versions
|
||||
prior to v1.8, described in [CVE-2024-2221](https://cve.mitre.org/cgi-bin/cvename.cgi?name=CVE-2024-2221).
|
||||
prior to v1.9, described in [CVE-2024-2221](https://cve.mitre.org/cgi-bin/cvename.cgi?name=CVE-2024-2221).
|
||||
The vulnerability allows an attacker to upload arbitrary files to the
|
||||
filesystem, which can be used to gain remote code execution.
|
||||
|
||||
@@ -31,35 +31,37 @@ filesystem is read-only and authentication is enabled by default. At worst,
|
||||
the vulnerability could be used by an authenticated user to crash a cluster,
|
||||
which is already possible, such as by uploading more vectors than can fit in RAM.
|
||||
|
||||
Qdrant has addressed the vulnerability in v1.8.3 and above with code that
|
||||
Qdrant has addressed the vulnerability in v1.9.0 and above with code that
|
||||
restricts file uploads to a folder dedicated to that purpose.
|
||||
|
||||
### Action
|
||||
|
||||
Check the current version of your Qdrant deployment. Upgrade if your deployment
|
||||
is not at least v1.8.3.
|
||||
is not at least v1.9.0.
|
||||
|
||||
To confirm the version of your Qdrant deployment in the cloud or on your local
|
||||
or cloud system, run an API GET call, as described in the [Qdrant Quickstart
|
||||
guide](https://qdrant.tech/documentation/cloud/quickstart-cloud/#step-2-test-cluster-access).
|
||||
guide](/documentation/cloud/quickstart-cloud/#step-2-test-cluster-access).
|
||||
If your Qdrant deployment is local, you do not need an API key.
|
||||
|
||||
Your next step depends on how you installed Qdrant. For details, read the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
[Qdrant Installation](/documentation/guides/installation/)
|
||||
guide.
|
||||
|
||||
#### If you use the Qdrant container or binary
|
||||
|
||||
Upgrade your deployment. Run the commands in the applicable section of the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
[Qdrant Installation](/documentation/guides/installation/)
|
||||
guide. The default commands automatically pull the latest version of Qdrant.
|
||||
|
||||
#### If you use the Qdrant helm chart
|
||||
|
||||
If you’ve set up Qdrant on kubernetes using a helm chart, follow the README in
|
||||
the [qdrant-helm](https://github.com/qdrant/qdrant-helm/tree/main?tab=readme-ov-file#upgrading) repository.
|
||||
Make sure applicable configuration files point to version v1.8.3 or above.
|
||||
Make sure applicable configuration files point to version v1.9.0 or above.
|
||||
|
||||
#### If you use the Qdrant cloud
|
||||
|
||||
No action is required. This vulnerability does not materially affect you. However, we suggest that you upgrade your cloud deployment to the latest version.
|
||||
|
||||
> Note: This article has been updated on 2024-05-10 to encourage users to upgrade to 1.9.0 to ensure protection from both CVE-2024-2221 and CVE-2024-3829.
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
---
|
||||
title: "Response to CVE-2024-3829: Arbitrary file upload vulnerability"
|
||||
draft: false
|
||||
slug: cve-2024-3829-response
|
||||
short_description: Qdrant keeps your systems secure
|
||||
description: Upgrade your deployments to at least v1.9.0. Cloud deployments not materially affected.
|
||||
preview_image: /blog/cve-2024-3829-response/cve-2024-3829-response-social-preview.png
|
||||
|
||||
# social_preview_image: /blog/Article-Image.png # Optional image used for link previews
|
||||
# title_preview_image: /blog/Article-Image.png # Optional image used for blog post title
|
||||
# small_preview_image: /blog/Article-Image.png # Optional image used for small preview in the list of blog posts
|
||||
date: 2024-06-10T17:00:00Z
|
||||
author: Mac Chaffee
|
||||
featured: false
|
||||
tags:
|
||||
- cve
|
||||
- security
|
||||
weight: 0 # Change this weight to change order of posts
|
||||
# For more guidance, see https://github.com/qdrant/landing_page?tab=readme-ov-file#blog
|
||||
---
|
||||
|
||||
### Summary
|
||||
|
||||
A security vulnerability has been discovered in Qdrant affecting all versions
|
||||
prior to v1.9, described in [CVE-2024-3829](https://cve.mitre.org/cgi-bin/cvename.cgi?name=CVE-2024-3829).
|
||||
The vulnerability allows an attacker to upload arbitrary files to the
|
||||
filesystem, which can be used to gain remote code execution. This is a different but similar vulnerability to CVE-2024-2221, announced in April 2024.
|
||||
|
||||
The vulnerability does not materially affect Qdrant cloud deployments, as that
|
||||
filesystem is read-only and authentication is enabled by default. At worst,
|
||||
the vulnerability could be used by an authenticated user to crash a cluster,
|
||||
which is already possible, such as by uploading more vectors than can fit in RAM.
|
||||
|
||||
Qdrant has addressed the vulnerability in v1.9.0 and above with code that
|
||||
restricts file uploads to a folder dedicated to that purpose.
|
||||
|
||||
### Action
|
||||
|
||||
Check the current version of your Qdrant deployment. Upgrade if your deployment
|
||||
is not at least v1.9.0.
|
||||
|
||||
To confirm the version of your Qdrant deployment in the cloud or on your local
|
||||
or cloud system, run an API GET call, as described in the [Qdrant Quickstart
|
||||
guide](https://qdrant.tech/documentation/cloud/quickstart-cloud/#step-2-test-cluster-access).
|
||||
If your Qdrant deployment is local, you do not need an API key.
|
||||
|
||||
Your next step depends on how you installed Qdrant. For details, read the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
guide.
|
||||
|
||||
#### If you use the Qdrant container or binary
|
||||
|
||||
Upgrade your deployment. Run the commands in the applicable section of the
|
||||
[Qdrant Installation](https://qdrant.tech/documentation/guides/installation/)
|
||||
guide. The default commands automatically pull the latest version of Qdrant.
|
||||
|
||||
#### If you use the Qdrant helm chart
|
||||
|
||||
If you’ve set up Qdrant on kubernetes using a helm chart, follow the README in
|
||||
the [qdrant-helm](https://github.com/qdrant/qdrant-helm/tree/main?tab=readme-ov-file#upgrading) repository.
|
||||
Make sure applicable configuration files point to version v1.9.0 or above.
|
||||
|
||||
#### If you use the Qdrant cloud
|
||||
|
||||
No action is required. This vulnerability does not materially affect you. However, we suggest that you upgrade your cloud deployment to the latest version.
|
||||
@@ -63,5 +63,4 @@ Thanks to the [DataTalks.Club](https://datatalks.club) for organizing [this podc
|
||||
If you're interested in a similar discussion, watch for the recording from the [following event](https://www.eventbrite.co.uk/e/the-evolution-of-genai-exploring-practical-applications-tickets-778359172237?aff=oddtdtcreator), organized by [DeepRec.ai](https://deeprec.ai).
|
||||
|
||||
### Further reading
|
||||
- https://qdrant.tech/blog
|
||||
- https://hub.superlinked.com/blog
|
||||
- [Qdrant Blog](/blog/)
|
||||
|
||||
@@ -0,0 +1,388 @@
|
||||
---
|
||||
title: "DSPy vs LangChain: A Comprehensive Framework Comparison" #required
|
||||
short_description: DSPy and LangChain are powerful frameworks for building AI applications leveraging LLMs and vector search technology.
|
||||
description: We dive deep into the capabilities of DSPy and LangChain and discuss scenarios where each of these frameworks shine. #required
|
||||
social_preview_image: /blog/dspy-vs-langchain/dspy-langchain.png # This image will be used in
|
||||
preview_image: /blog/dspy-vs-langchain/dspy-langchain.png
|
||||
author: Qdrant Team # Author of the article. Required.
|
||||
author_link: https://qdrant.tech/ # Link to the author's page. Required.
|
||||
date: 2024-02-23T08:00:00-03:00 # Date of the article. Required.
|
||||
draft: false # If true, the article will not be published
|
||||
keywords: # Keywords for SEO
|
||||
- DSPy
|
||||
- LangChain
|
||||
- AI frameworks
|
||||
- LLMs
|
||||
- vector search
|
||||
- RAG applications
|
||||
- chatbots
|
||||
---
|
||||
|
||||
As Large Language Models (LLMs) and vector stores have become steadily more powerful, a new generation of frameworks has appeared which can streamline the development of AI applications by leveraging LLMs and vector search technology. These frameworks simplify the process of building everything from Retrieval Augmented Generation (RAG) applications to complex chatbots with advanced conversational abilities, and even sophisticated reasoning-driven AI applications.
|
||||
|
||||
The most well-known of these frameworks is possibly [LangChain](https://github.com/langchain-ai/langchain). [Launched in October 2022](https://en.wikipedia.org/wiki/LangChain) as an open-source project by Harrison Chase, the project quickly gained popularity, attracting contributions from hundreds of developers on GitHub. LangChain excels in its broad support for documents, data sources, and APIs. This, along with seamless integration with vector stores like Qdrant and the ability to chain multiple LLMs, has allowed developers to build complex AI applications without reinventing the wheel.
|
||||
|
||||
However, despite the many capabilities unlocked by frameworks like LangChain, developers still needed expertise in [prompt engineering](https://en.wikipedia.org/wiki/Prompt_engineering) to craft optimal LLM prompts. Additionally, optimizing these prompts and adapting them to build multi-stage reasoning AI remained challenging with the existing frameworks.
|
||||
|
||||
In fact, as you start building production-grade AI applications, it becomes clear that a single LLM call isn’t enough to unlock the full capabilities of LLMs. Instead, you need to create a workflow where the model interacts with external tools like web browsers, fetches relevant snippets from documents, and compiles the results into a multi-stage reasoning pipeline.
|
||||
|
||||
This involves building an architecture that combines and reasons on intermediate outputs, with LLM prompts that adapt according to the task at hand, before producing a final output. A manual approach to prompt engineering quickly falls short in such scenarios.
|
||||
|
||||
In October 2023, researchers working in Stanford NLP released a library, [DSPy](https://github.com/stanfordnlp/dspy), which entirely automates the process of optimizing prompts and weights for large language models (LLMs), eliminating the need for manual prompting or prompt engineering.
|
||||
|
||||
One of DSPy's key features is its ability to automatically tune LLM prompts, an approach that is especially powerful when your application needs to call the LLM several times within a pipeline.
|
||||
|
||||
So, when building an LLM and vector store-backed AI application, which of these frameworks should you choose? In this article, we dive deep into the capabilities of each and discuss scenarios where each of these frameworks shine. Let’s get started!
|
||||
|
||||
## **LangChain: Features, Performance, and Use Cases**
|
||||
|
||||
LangChain, as discussed above, is an open-source orchestration framework available in both [Python](https://python.langchain.com/v0.2/docs/introduction/) and [JavaScript](https://js.langchain.com/v0.2/docs/introduction/), designed to simplify the development of AI applications leveraging LLMs. For developers working with one or multiple LLMs, it acts as a universal interface for these AI models. LangChain integrates with various external data sources, supports a wide range of data types and stores, streamlines the handling of vector embeddings and retrieval through similarity search, and simplifies the integration of AI applications with existing software workflows.
|
||||
|
||||
At a high level, LangChain abstracts the common steps required to work with language models into modular components, which serve as the building blocks of AI applications. These components can be "chained" together to create complex applications. Thanks to these abstractions, LangChain allows for rapid experimentation and prototyping of AI applications in a short timeframe.
|
||||
|
||||
LangChain breaks down the functionality required to build AI applications into three key sections:
|
||||
|
||||
- **Model I/O**: Building blocks to interface with the LLM.
|
||||
- **Retrieval**: Building blocks to streamline the retrieval of data used by the LLM for generation (such as the retrieval step in RAG applications).
|
||||
- **Composition**: Components to combine external APIs, services and other LangChain primitives.
|
||||
|
||||
These components are pulled together into ‘chains’ that are constructed using [LangChain Expression Language](https://python.langchain.com/v0.1/docs/expression_language/) (LCEL). We’ill first look at the various building blocks, and then see how they can be combined using LCEL.
|
||||
|
||||
### **LLM Model I/O**
|
||||
|
||||
LangChain offers broad compatibility with various LLMs, and its [LLM](https://python.langchain.com/v0.1/docs/modules/model_io/llms/) class provides a standard interface to these models. Leveraging proprietary models offered by platforms like OpenAI, Mistral, Cohere, or Gemini is straightforward and requires just an API key from the respective platform.
|
||||
|
||||
For instance, to use OpenAI models, you simply need to do the following:
|
||||
|
||||
```python
|
||||
from langchain_openai import OpenAI
|
||||
|
||||
llm = OpenAI(api_key="...")
|
||||
|
||||
llm.invoke("Where is Paris?")
|
||||
|
||||
```
|
||||
|
||||
|
||||
Open-source models like Meta AI’s Llama variants (such as Llama3-8B) or Mistral AI’s open models (like Mistral-7B) can be easily integrated using their Hugging Face endpoints or local LLM deployment tools like Ollama, vLLM, or LM Studio. You can also use the [CustomLLM](https://python.langchain.com/v0.1/docs/modules/model_io/llms/custom_llm/) class to build Custom LLM wrappers.
|
||||
|
||||
Here’s how simple it is to use LangChain with LlaMa3-8B, using [Ollama](https://ollama.com/).
|
||||
|
||||
```python
|
||||
from langchain_community.llms import Ollama
|
||||
|
||||
llm = Ollama(model="llama3")
|
||||
|
||||
llm.invoke("Where is Berlin?")
|
||||
|
||||
```
|
||||
|
||||
|
||||
LangChain also offers output parsers to structure the LLM output in a format that the application may need, such as structured data types like JSON, XML, CSV, and others. To understand LangChain’s interface with LLMs in detail, read the documentation [here](https://python.langchain.com/v0.1/docs/modules/model_io/).
|
||||
|
||||
### **Retrieval**
|
||||
|
||||
Most enterprise AI applications are built by augmenting the LLM context using data specific to the application’s use case. To accomplish this, the relevant data needs to be first retrieved, typically using vector similarity search, and then passed to the LLM context at the generation step. This architecture, known as [Retrieval Augmented Generation](/articles/what-is-rag-in-ai/) (RAG), can be used to build a wide range of AI applications.
|
||||
|
||||
While the retrieval process sounds simple, it involves a number of complex steps: loading data from a source, splitting it into chunks, converting it into vectors or vector embeddings, storing it in a vector store, and then retrieving results based on a query before the generation step.
|
||||
|
||||
LangChain offers a number of building blocks to make this retrieval process simpler.
|
||||
|
||||
- **Document Loaders**: LangChain offers over 100 different document loaders, including integrations with providers like Unstructured or Airbyte. It also supports loading various types of documents, such as PDFs, HTML, CSV, and code, from a range of locations like S3.
|
||||
- **Splitting**: During the retrieval step, you typically need to retrieve only the relevant section of a document. To do this, you need to split a large document into smaller chunks. LangChain offers various document transformers that make it easy to split, combine, filter, or manipulate documents.
|
||||
- **Text Embeddings**: A key aspect of the retrieval step is converting document chunks into vectors, which are high-dimensional numerical representations that capture the semantic meaning of the text. LangChain offers integrations with over 25 embedding providers and methods, such as [FastEmbed](https://github.com/qdrant/fastembed).
|
||||
- **Vector Store Integration**: LangChain integrates with over 50 vector stores, including specialized ones like [Qdrant](/documentation/frameworks/langchain/), and exposes a standard interface.
|
||||
- **Retrievers**: LangChain offers various retrieval algorithms and allows you to use third-party retrieval algorithms or create custom retrievers.
|
||||
- **Indexing**: LangChain also offers an indexing API that keeps data from any data source in sync with the vector store, helping to reduce complexities around managing unchanged content or avoiding duplicate content.
|
||||
|
||||
### **Composition**
|
||||
|
||||
Finally, LangChain also offers building blocks that help combine external APIs, services, and LangChain primitives. For instance, it provides tools to fetch data from Wikipedia or search using Google Lens. The list of tools it offers is [extremely varied](https://python.langchain.com/v0.1/docs/integrations/tools/).
|
||||
|
||||
LangChain also offers ways to build agents that use language models to decide on the sequence of actions to take.
|
||||
|
||||
### **LCEL**
|
||||
|
||||
The primary method of building an application in LangChain is through the use of [LCEL](https://python.langchain.com/v0.1/docs/expression_language/), the LangChain Expression Language. It is a declarative syntax designed to simplify the composition of chains within the LangChain framework. It provides a minimalist code layer that enables the rapid development of chains, leveraging advanced features such as streaming, asynchronous execution, and parallel processing.
|
||||
|
||||
LCEL is particularly useful for building chains that involve multiple language model calls, data transformations, and the integration of outputs from language models into downstream applications.
|
||||
|
||||
### **Some Use Cases of LangChain**
|
||||
|
||||
Given the flexibility that LangChain offers, a wide range of applications can be built using the framework. Here are some examples:
|
||||
|
||||
**RAG Applications**: LangChain provides all the essential building blocks needed to build Retrieval Augmented Generation (RAG) applications. It integrates with vector stores and LLMs, streamlining the entire process of loading, chunking, and retrieving relevant sections of a document in a few lines of code.
|
||||
|
||||
**Chatbots**: LangChain offers a suite of components that streamline the process of building conversational chatbots. These include chat models, which are specifically designed for message-based interactions and provide a conversational tone suitable for chatbots.
|
||||
|
||||
**Extracting Structured Outputs**: LangChain assists in extracting structured output from data using various tools and methods. It supports multiple extraction approaches, including tool/function calling mode, JSON mode, and prompting-based extraction.
|
||||
|
||||
**Agents**: LangChain simplifies the process of building agents by providing building blocks and integration with LLMs, enabling developers to construct complex, multi-step workflows. These agents can interact with external data sources and tools, and generate dynamic and context-aware responses for various applications.
|
||||
|
||||
If LangChain offers such a wide range of integrations and the primary building blocks needed to build AI applications, *why do we need another framework?*
|
||||
|
||||
As Omar Khattab, PhD, Stanford and researcher at Stanford NLP, said when introducing DSPy in his [talk](https://www.youtube.com/watch?v=Dt3H2ninoeY) at ‘Scale By the Bay’ in November 2023: “We can build good reliable systems with these new artifacts that are language models (LMs), but importantly, this is conditioned on us *adapting* them as well as *stacking* them well”.
|
||||
|
||||
## **DSPy: Features, Performance, and Use Cases**
|
||||
|
||||
When building AI systems, developers need to break down the task into multiple reasoning steps, adapt language model (LM) prompts for each step until they get the right results, and then ensure that the steps work together to achieve the desired outcome.
|
||||
|
||||
Complex multihop pipelines, where multiple LLM calls are stacked, are messy. They involve string-based prompting tricks or prompt hacks at each step, and getting the pipeline to work is even trickier.
|
||||
|
||||
Additionally, the manual prompting approach is highly unscalable, as any change in the underlying language model breaks the prompts and the pipeline. LMs are highly sensitive to prompts and slight changes in wording, context, or phrasing can significantly impact the model's output. Due to this, despite the functionality provided by frameworks like LangChain, developers often have to spend a lot of time engineering prompts to get the right results from LLMs.
|
||||
|
||||
How do you build a system that’s less brittle and more predictable? Enter DSPy!
|
||||
|
||||
[DSPy](https://github.com/stanfordnlp/dspy) is built on the paradigm that language models (LMs) should be programmed rather than prompted. The framework is designed for algorithmically optimizing and adapting LM prompts and weights, and focuses on replacing prompting techniques with a programming-centric approach.
|
||||
|
||||
DSPy treats the LM like a device and abstracts out the underlying complexities of prompting. To achieve this, DSPy introduces three simple building blocks:
|
||||
|
||||
### **Signatures**
|
||||
|
||||
[Signatures](https://dspy-docs.vercel.app/docs/building-blocks/signatures) replace handwritten prompts and are written in natural language. They are simply declarations or specs of the behavior that you expect from the language model. Some examples are:
|
||||
|
||||
- question -> answer
|
||||
- long_document -> summary
|
||||
- context, question -> rationale, response
|
||||
|
||||
Rather than manually crafting complex prompts or engaging in extensive fine-tuning of LLMs, signatures allow for the automatic generation of optimized prompts.
|
||||
|
||||
DSPy Signatures can be specified in two ways:
|
||||
|
||||
1. Inline Signatures: Simple tasks can be defined in a concise format, like "question -> answer" for question-answering or "document -> summary" for summarization.
|
||||
|
||||
2. Class-Based Signatures: More complex tasks might require class-based signatures, which can include additional instructions or descriptions about the inputs and outputs. For example, a class for emotion classification might clearly specify the range of emotions that can be classified.
|
||||
|
||||
### **Modules**
|
||||
|
||||
Modules take signatures as input, and automatically generate high-quality prompts. Inspired heavily from PyTorch, DSPy [modules](https://dspy-docs.vercel.app/docs/building-blocks/modules) eliminate the need for crafting prompts manually.
|
||||
|
||||
The framework supports advanced modules like [dspy.ChainOfThought](https://dspy-docs.vercel.app/api/modules/ChainOfThought), which adds step-by-step rationalization before producing an output. The output not only provides answers but also rationales. Other modules include [dspy.ProgramOfThought](https://dspy-docs.vercel.app/api/modules/ProgramOfThought), which outputs code whose execution results dictate the response, and [dspy.ReAct](https://dspy-docs.vercel.app/api/modules/ReAct), an agent that uses tools to implement signatures.
|
||||
|
||||
DSPy also offers modules like [dspy.MultiChainComparison](https://dspy-docs.vercel.app/api/modules/MultiChainComparison), which can compare multiple outputs from dspy.ChainOfThought in order to produce a final prediction. There are also utility modules like [dspy.majority](https://dspy-docs.vercel.app/docs/building-blocks/modules#what-other-dspy-modules-are-there-how-can-i-use-them) for aggregating responses through voting.
|
||||
|
||||
Modules can be composed into larger programs, and you can compose multiple modules into bigger modules. This allows you to create complex, behavior-rich applications using language models.
|
||||
|
||||
### **Optimizers**
|
||||
|
||||
[Optimizers](https://dspy-docs.vercel.app/docs/building-blocks/optimizers) take a set of modules that have been connected to create a pipeline, compile them into auto-optimized prompts, and maximize an outcome metric.
|
||||
|
||||
Essentially, optimizers are designed to generate, test, and refine prompts, and ensure that the final prompt is highly optimized for the specific dataset and task at hand. Using optimizers in the DSPy framework significantly simplifies the process of developing and refining LM applications by automating the prompt engineering process.
|
||||
|
||||
### **Building AI Applications with DSPy**
|
||||
|
||||
A typical DSPy program requires the developer to follow the following 8 steps:
|
||||
|
||||
1. **Defining the Task**: Identify the specific problem you want to solve, including the input and output formats.
|
||||
2. **Defining the Pipeline**: Plan the sequence of operations needed to solve the task. Then craft the signatures and the modules.
|
||||
3. **Testing with Examples**: Run the pipeline with a few examples to understand the initial performance. This helps in identifying immediate issues with the program and areas for improvement.
|
||||
4. **Defining Your Data**: Prepare and structure your training and validation datasets. This is needed by the optimizer for training the model and evaluating its performance accurately.
|
||||
5. **Defining Your Metric**: Choose metrics that will measure the success of your model. These metrics help the optimizer evaluate how well the model is performing.
|
||||
6. **Collecting Zero-Shot Evaluations**: Run initial evaluations without prior training to establish a baseline. This helps in understanding the model’s capabilities and limitations out of the box.
|
||||
7. **Compiling with a DSPy Optimizer**: Given the data and metric, you can now optimize the program. DSPy offers a variety of optimizers designed for different purposes. These optimizers can generate step-by-step examples, craft detailed instructions, and/or update language model prompts and weights as needed.
|
||||
8. **Iterating**: Continuously refine each aspect of your task, from the pipeline and data to the metrics and evaluations. Iteration helps in gradually improving the model’s performance and adapting to new requirements.
|
||||
9.
|
||||
|
||||
|
||||
{{< figure src=/blog/dspy-vs-langchain/process.jpg caption="Process" >}}
|
||||
|
||||
**Language Model Setup**
|
||||
|
||||
Setting up the LM in DSPy is easy.
|
||||
|
||||
```python
|
||||
# pip install dspy
|
||||
|
||||
import dspy
|
||||
|
||||
llm = dspy.OpenAI(model='gpt-3.5-turbo-1106', max_tokens=300)
|
||||
|
||||
dspy.configure(lm=llm)
|
||||
|
||||
# Let's test this. First define a module (ChainOfThought) and assign it a signature (return an answer, given a question).
|
||||
|
||||
qa = dspy.ChainOfThought('question -> answer')
|
||||
|
||||
# Then, run with the default LM configured.
|
||||
|
||||
response = qa(question="Where is Paris?")
|
||||
|
||||
print(response.answer)
|
||||
|
||||
```
|
||||
|
||||
You are not restricted to using one LLM in your program; you can use [multiple](https://dspy-docs.vercel.app/docs/building-blocks/language_models#using-multiple-lms-at-once). DSPy can be used with both managed models such as OpenAI, Cohere, Anyscale, Together, or PremAI as well as with local LLM deployments through vLLM, Ollama, or TGI server. All LLM calls are cached by default.
|
||||
|
||||
**Vector Store Integration (Retrieval Model)**
|
||||
|
||||
You can easily set up [Qdrant](/documentation/frameworks/dspy/) vector store to act as the retrieval model. To do so, follow these steps:
|
||||
|
||||
```python
|
||||
# pip install dspy-ai[qdrant]
|
||||
|
||||
import dspy
|
||||
|
||||
from dspy.retrieve.qdrant_rm import QdrantRM
|
||||
|
||||
from qdrant_client import QdrantClient
|
||||
|
||||
llm = dspy.OpenAI(model="gpt-3.5-turbo")
|
||||
|
||||
qdrant_client = QdrantClient()
|
||||
|
||||
qdrant_rm = QdrantRM("collection-name", qdrant_client, k=3)
|
||||
|
||||
dspy.settings.configure(lm=llm, rm=qdrant_rm)
|
||||
|
||||
```
|
||||
|
||||
The above code sets up DSPy to use Qdrant (localhost), with collection-name as the default retrieval client. You can now build a RAG module in the following way:
|
||||
|
||||
```python
|
||||
|
||||
class RAG(dspy.Module):
|
||||
def __init__(self, num_passages=5):
|
||||
super().__init__()
|
||||
|
||||
self.retrieve = dspy.Retrieve(k=num_passages)
|
||||
self.generate_answer = dspy.ChainOfThought('context, question -> answer') # using inline signature
|
||||
|
||||
def forward(self, question):
|
||||
context = self.retrieve(question).passages
|
||||
prediction = self.generate_answer(context=context, question=question)
|
||||
return dspy.Prediction(context=context, answer=prediction.answer)
|
||||
|
||||
```
|
||||
|
||||
Now you can use the RAG module like any Python module.
|
||||
|
||||
**Optimizing the Pipeline**
|
||||
|
||||
In this step, DSPy requires you to create a training dataset and a metric function, which can help validate the output of your program. Using this, DSPy tunes the parameters (i.e., the prompts and/or the LM weights) to maximize the accuracy of the RAG pipeline.
|
||||
|
||||
Using DSPy optimizers involves the following steps:
|
||||
|
||||
1. Set up your DSPy program with the desired signatures and modules.
|
||||
2. Create a training and validation dataset, with example input and output that you expect from your DSPy program.
|
||||
3. Choose an appropriate optimizer such as BootstrapFewShotWithRandomSearch, MIPRO, or BootstrapFinetune.
|
||||
4. Create a metric function that evaluates the performance of the DSPy program. You can evaluate based on accuracy or quality of responses, or on a metric that’s relevant to your program.
|
||||
5. Run the optimizer with the DSPy program, metric function, and training inputs. DSPy will compile the program and automatically adjust parameters and improve performance.
|
||||
6. Use the compiled program to perform the task. Iterate and adapt if required.
|
||||
|
||||
To learn more about optimizing DSPy programs, read [this](https://dspy-docs.vercel.app/docs/building-blocks/optimizers).
|
||||
|
||||
DSPy is heavily influenced by PyTorch, and replaces complex prompting with reusable modules for common tasks. Instead of crafting specific prompts, you write code that DSPy automatically translates for the LLM. This, along with built-in optimizers, makes working with LLMs more systematic and efficient.
|
||||
|
||||
### **Use Cases of DSPy**
|
||||
|
||||
As we saw above, DSPy can be used to create fairly complex applications which require stacking multiple LM calls without the need for prompt engineering. Even though the framework is comparatively new - it started gaining popularity since November 2023 when it was first introduced - it has created a promising new direction for LLM-based applications.
|
||||
|
||||
Here are some of the possible uses of DSPy:
|
||||
|
||||
**Automating Prompt Engineering**: DSPy automates the process of creating prompts for LLMs, and allows developers to focus on the core logic of their application. This is powerful as manual prompt engineering makes AI applications highly unscalable and brittle.
|
||||
|
||||
**Building Chatbots**: The modular design of DSPy makes it well-suited for creating chatbots with improved response quality and faster development cycles. DSPy's automatic prompting and optimizers can help ensure chatbots generate consistent and informative responses across different conversation contexts.
|
||||
|
||||
**Complex Information Retrieval Systems**: DSPy programs can be easily integrated with vector stores, and used to build multi-step information retrieval systems with stacked calls to the LLM. This can be used to build highly sophisticated retrieval systems. For example, DSPy can be used to develop custom search engines that understand complex user queries and retrieve the most relevant information from vector stores.
|
||||
|
||||
**Improving LLM Pipelines**: One of the best uses of DSPy is to optimize LLM pipelines. DSPy's modular design greatly simplifies the integration of LLMs into existing workflows. Additionally, DSPy's built-in optimizers can help fine-tune LLM pipelines based on desired metrics.
|
||||
|
||||
**Multi-Hop Question-Answering**: Multi-hop question-answering involves answering complex questions that require reasoning over multiple pieces of information, which are often scattered across different documents or sections of text. With DSPy, users can leverage its automated prompt engineering capabilities to develop prompts that effectively guide the model on how to piece together information from various sources.
|
||||
|
||||
## **Comparative Analysis: DSPy vs LangChain**
|
||||
|
||||
DSPy and LangChain are both powerful frameworks for building AI applications, leveraging large language models (LLMs) and vector search technology. Below is a comparative analysis of their key features, performance, and use cases:
|
||||
|
||||
| Feature | LangChain | DSPy |
|
||||
| --- | --- | --- |
|
||||
| Core Focus | Focus on providing a large number of building blocks to simplify the development of applications that use LLMs in conjunction with user-specified data sources. | Focus on automating and modularizing LLM interactions, eliminating manual prompt engineering and improving systematic reliability. |
|
||||
| Approach | Utilizes modular components and chains that can be linked together using the LangChain Expression Language (LCEL). | Streamlines LLM interaction by prioritizing programming instead of prompting, and automating prompt refinement and weight tuning. |
|
||||
| Complex Pipelines | Facilitates the creation of chains using LCEL, supporting asynchronous execution and integration with various data sources and APIs. | Simplifies multi-stage reasoning pipelines using modules and optimizers, and ensures scalability through less manual intervention. |
|
||||
| Optimization | Relies on user expertise for prompt engineering and chaining of multiple LLM calls. | Includes built-in optimizers that automatically tune prompts and weights, and helps bring efficiency and effectiveness in LLM pipelines. |
|
||||
| Community and Support | Large open-source community with extensive documentation and examples. | Emerging framework with growing community support, and bringing a paradigm-shift in LLM prompting. |
|
||||
|
||||
### **LangChain**
|
||||
|
||||
Strengths:
|
||||
|
||||
1. Data Sources and APIs: LangChain supports a wide variety of data sources and APIs, and allows seamless integration with different types of data. This makes it highly versatile for various AI applications.
|
||||
2. LangChain provides modular components that can be chained together and allows you to create complex AI workflows. LangChain Expression Language (LCEL) lets you use declarative syntax and makes it easier to build and manage workflows.
|
||||
3. Since LangChain is an older framework, it has extensive documentation and thousands of examples that developers can take inspiration from.
|
||||
|
||||
Weaknesses:
|
||||
|
||||
1. For projects involving complex, multi-stage reasoning tasks, LangChain requires significant manual prompt engineering. This can be time-consuming and prone to errors.
|
||||
2. Scalability Issues: Managing and scaling workflows that require multiple LLM calls can be pretty challenging.
|
||||
3. Developers need sound understanding of prompt engineering in order to build applications that require multiple calls to the LLM.
|
||||
|
||||
### **DSPy**
|
||||
|
||||
Strengths:
|
||||
|
||||
1. DSPy automates the process of prompt generation and optimization, and significantly reduces the need for manual prompt engineering. This makes working with LLMs easier and helps build scalable AI workflows.
|
||||
2. The framework includes built-in optimizers like BootstrapFewShot and MIPRO, which automatically refine prompts and adapt them to specific datasets.
|
||||
3. DSPy uses general-purpose modules and optimizers to simplify the complexities of prompt engineering. This can help you create complex multi-step reasoning applications easily, without worrying about the intricacies of dealing with LLMs.
|
||||
4. DSPy supports various LLMs, including the flexibility of using multiple LLMs in the same program.
|
||||
5. By focusing on programming rather than prompting, DSPy ensures higher reliability and performance for AI applications, particularly those that require complex multi-stage reasoning.
|
||||
|
||||
Weaknesses:
|
||||
|
||||
1. As a newer framework, DSPy has a smaller community compared to LangChain. This means you will have limited availability of resources, examples, and community support.
|
||||
2. Although DSPy offers tutorials and guides, its documentation is less extensive than LangChain’s, which can pose challenges when you start.
|
||||
3. When starting with DSPy, you may feel limited to the paradigms and modules it provides.
|
||||
|
||||
## **Selecting the Ideal Framework for Your AI Project**
|
||||
|
||||
When deciding between DSPy and LangChain for your AI project, you should consider the problem statement and choose the framework that best aligns with your project goals.
|
||||
|
||||
Here are some guidelines:
|
||||
|
||||
### **Project Type**
|
||||
|
||||
**LangChain**: LangChain is ideal for projects that require extensive integration with multiple data sources and APIs, especially projects that benefit from the wide range of document loaders, vector stores, and retrieval algorithms that it supports.
|
||||
|
||||
**DSPy**: DSPy is best suited for projects that involve complex multi-stage reasoning pipelines or those that may eventually need stacked LLM calls. DSPy’s systematic approach to prompt engineering and its ability to optimize LLM interactions can help create highly reliable AI applications.
|
||||
|
||||
### **Technical Expertise**
|
||||
|
||||
**LangChain**: As the complexity of the application grows, LangChain requires a good understanding of prompt engineering and expertise in chaining multiple LLM calls.
|
||||
|
||||
**DSPy**: Since DSPy is designed to abstract away the complexities of prompt engineering, it makes it easier for developers to focus on high-level logic rather than low-level prompt crafting.
|
||||
|
||||
### **Community and Support**
|
||||
|
||||
**LangChain**: LangChain boasts a large and active community with extensive documentation, examples, and active contributions, and you will find it easier to get going.
|
||||
|
||||
**DSPy**: Although newer and with a smaller community, DSPy is growing rapidly and offers tutorials and guides for some of the key use cases. DSPy may be more challenging to get started with, but its architecture makes it highly scalable.
|
||||
|
||||
### **Use Case Scenarios**
|
||||
|
||||
**Retrieval Augmented Generation (RAG) Applications**
|
||||
|
||||
**LangChain**: Excellent for building simple RAG applications due to its robust support for vector stores, document loaders, and retrieval algorithms.
|
||||
|
||||
**DSPy**: Suitable for RAG applications requiring high reliability and automated prompt optimization, ensuring consistent performance across complex retrieval tasks.
|
||||
|
||||
**Chatbots and Conversational AI**
|
||||
|
||||
**LangChain**: Provides a wide range of components for building conversational AI, making it easy to integrate LLMs with external APIs and services.
|
||||
|
||||
**DSPy**: Ideal for developing chatbots that need to handle complex, multi-stage conversations with high reliability and performance. DSPy’s automated optimizations ensure consistent and contextually accurate responses.
|
||||
|
||||
**Complex Information Retrieval Systems**
|
||||
|
||||
**LangChain**: Effective for projects that require seamless integration with various data sources and sophisticated retrieval capabilities.
|
||||
|
||||
**DSPy**: Best for systems that involve complex multi-step retrieval processes, where prompt optimization and modular design can significantly enhance performance and reliability.
|
||||
|
||||
You can also choose to combine and use the best features of both. In fact, LangChain has released an [integration with DSPy](https://python.langchain.com/v0.1/docs/integrations/providers/dspy/) to simplify this process. This allows you to use some of the utility functions that LangChain provides, such as text splitter, directory loaders, or integrations with other data sources while using DSPy for the LM interactions.
|
||||
|
||||
## **Level Up Your AI Projects with Advanced Frameworks**
|
||||
|
||||
LangChain and DSPy both offer unique capabilities and can help you build powerful AI applications. Qdrant integrates with both LangChain and DSPy, allowing you to leverage its performance, efficiency and security features in either scenario. LangChain is ideal for projects that require extensive integration with various data sources and APIs. On the other hand, DSPy offers a powerful paradigm for building complex multi-stage applications. For pulling together an AI application that doesn’t require much prompt engineering, use LangChain. However, pick DSPy when you need a systematic approach to prompt optimization and modular design, and need robustness and scalability for complex, multi-stage reasoning applications.
|
||||
|
||||
## **References**
|
||||
|
||||
[https://python.langchain.com/v0.1/docs/get_started/introduction](https://python.langchain.com/v0.1/docs/get_started/introduction)
|
||||
|
||||
[https://dspy-docs.vercel.app/docs/intro](https://dspy-docs.vercel.app/docs/intro)
|
||||
+10
-11
@@ -1,14 +1,11 @@
|
||||
---
|
||||
draft: false
|
||||
title: Gen AI and Vector Search - Iveta Lohovska | Vector Space Talks
|
||||
title: Iveta Lohovska on Gen AI and Vector Search | Qdrant
|
||||
slug: gen-ai-and-vector-search
|
||||
short_description: Iveta talks about the importance of trustworthy AI,
|
||||
particularly when implementing it within high-stakes enterprises like
|
||||
governments and security agencies
|
||||
description: Iveta Lohovska discusses the importance of explainability and
|
||||
transparency, discussing high-stakes use cases in sectors like cybersecurity
|
||||
and climate data, and emphasizing the necessity for on-prem solutions and
|
||||
traceable vector databases to ensure data integrity and confidentiality.
|
||||
description: Discover valuable insights on generative AI, vector search, and ethical AI implementation from Iveta Lohovska, Chief Technologist at HPE.
|
||||
preview_image: /blog/from_cms/iveta-lohovska-bp-cropped.png
|
||||
date: 2024-04-11T22:12:00.000Z
|
||||
author: Demetrios Brinkmann
|
||||
@@ -19,11 +16,13 @@ tags:
|
||||
- Retrieval Augmented Generation
|
||||
- GenAI
|
||||
---
|
||||
# Exploring Gen AI and Vector Search: Insights from Iveta Lohovska
|
||||
|
||||
> *"In the generative AI context of AI, all foundational models have been trained on some foundational data sets that are distributed in different ways. Some are very conversational, some are very technical, some are on, let's say very strict taxonomy like healthcare or chemical structures. We call them modalities, and they have different representations.”*\
|
||||
— Iveta Lohovska
|
||||
>
|
||||
|
||||
Iveta Lohovska serves as the Chief Technologist and Principal Data Scientist for AI and Supercomputing at Hewlett Packard Enterprise (HPE), where she champions the democratization of decision intelligence and the development of ethical AI solutions. An industry leader, her multifaceted expertise encompasses natural language processing, computer vision, and data mining. Committed to leveraging technology for societal benefit, Iveta is a distinguished technical advisor to the United Nations' AI for Good program and a Data Science lecturer at the Vienna University of Applied Sciences. Her career also includes impactful roles with the World Bank Group, focusing on open data initiatives and Sustainable Development Goals (SDGs), as well as collaborations with USAID and the Gates Foundation.
|
||||
Iveta Lohovska serves as the Chief Technologist and Principal Data Scientist for AI and Supercomputing at [Hewlett Packard Enterprise (HPE)](https://www.hpe.com/us/en/home.html), where she champions the democratization of decision intelligence and the development of ethical AI solutions. An industry leader, her multifaceted expertise encompasses natural language processing, computer vision, and data mining. Committed to leveraging technology for societal benefit, Iveta is a distinguished technical advisor to the United Nations' AI for Good program and a Data Science lecturer at the Vienna University of Applied Sciences. Her career also includes impactful roles with the World Bank Group, focusing on open data initiatives and Sustainable Development Goals (SDGs), as well as collaborations with USAID and the Gates Foundation.
|
||||
|
||||
***Listen to the episode on [Spotify](https://open.spotify.com/episode/7f1RDwp5l2Ps9N7gKubl8S?si=kCSX4HGCR12-5emokZbRfw), Apple Podcast, Podcast addicts, Castbox. You can also watch this episode on [YouTube](https://youtu.be/RsRAUO-fNaA).***
|
||||
|
||||
@@ -33,7 +32,7 @@ Iveta Lohovska serves as the Chief Technologist and Principal Data Scientist for
|
||||
|
||||
## **Top takeaways:**
|
||||
|
||||
In our continuous pursuit of knowledge and understanding, especially in the evolving landscape of AI and the vector space, we brought another great Vector Space Talk episode featuring Iveta Lohovska as she talks about generative AI and vector search.
|
||||
In our continuous pursuit of knowledge and understanding, especially in the evolving landscape of AI and the vector space, we brought another great Vector Space Talk episode featuring Iveta Lohovska as she talks about generative AI and [vector search](https://qdrant.tech/).
|
||||
|
||||
Iveta brings valuable insights from her work with the World Bank and as Chief Technologist at HPE, explaining the ins and outs of ethical AI implementation.
|
||||
|
||||
@@ -101,7 +100,7 @@ Sabrina Aquino:
|
||||
That's amazing. And can you talk a little bit more about the importance of the transparency of these models and what can happen if we don't know exactly what kind of data they are being trained on?
|
||||
|
||||
Iveta Lohovska:
|
||||
I mean, this is especially relevant under our context of vector databases and vector search. Because in the generative AI context of AI, all foundational models have been trained on some foundational data sets that are distributed in different ways. Some are very conversational, some are very technical, some are on, let's say very strict taxonomy like healthcare or chemical structures. We call them modalities, and they have different representations. So, so when it comes to implementing vector search or vector database and knowing the distribution of the foundational data sets, you have better control if you introduce additional layers or additional components to have the control in your hands of where the information is coming from, where it's stored, what are the embeddings. So that helps, but it is actually quite important that you know what the foundational data sets are, so that you can predict any kind of weaknesses or vulnerabilities or penetrations that the solution or the use case of the model will face when it lands at the end user. Because we know with generative AI that is unpredictable, we know we can implement guardrails. They're already solutions.
|
||||
I mean, this is especially relevant under our context of [vector databases](https://qdrant.tech/articles/what-is-a-vector-database/) and vector search. Because in the generative AI context of AI, all foundational models have been trained on some foundational data sets that are distributed in different ways. Some are very conversational, some are very technical, some are on, let's say very strict taxonomy like healthcare or chemical structures. We call them modalities, and they have different representations. So, so when it comes to implementing vector search or [vector database](https://qdrant.tech/articles/what-is-a-vector-database/) and knowing the distribution of the foundational data sets, you have better control if you introduce additional layers or additional components to have the control in your hands of where the information is coming from, where it's stored, [what are the embeddings](https://qdrant.tech/articles/what-are-embeddings/). So that helps, but it is actually quite important that you know what the foundational data sets are, so that you can predict any kind of weaknesses or vulnerabilities or penetrations that the solution or the use case of the model will face when it lands at the end user. Because we know with generative AI that is unpredictable, we know we can implement guardrails. They're already solutions.
|
||||
|
||||
Iveta Lohovska:
|
||||
We know they're not 100, they don't give you 100% certainty, but they are definitely use cases and work where you need to hit the hundred percent certainty, especially intelligence, cybersecurity and healthcare.
|
||||
@@ -119,10 +118,10 @@ Iveta Lohovska:
|
||||
It has the public. You can go and benchmark your carbon footprint as an individual living in one country comparing to an individual living in another. But if you are a policymaker, which is the other interface of this application, who will write the policy recommendation of a country in their own country, or a country they're advising on, you might want to make sure that the scientific citations and the policy recommendations that you're making are correct and they are retrieved from the proper data sources. Because there will be a huge implication when you go public with those numbers or when you actually design a law that is reinforceable with legal terms and law enforcement.
|
||||
|
||||
Sabrina Aquino:
|
||||
That's very interesting, Iveta, and I think this is one of the great use cases for RAG, for example. And I think if you can talk a little bit more about how vector search is playing into all of this, how it's helping organizations do this, this.
|
||||
That's very interesting, Iveta, and I think this is one of the great use cases for [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/), for example. And I think if you can talk a little bit more about how vector search is playing into all of this, how it's helping organizations do this, this.
|
||||
|
||||
Iveta Lohovska:
|
||||
Would be amazing in such specific use cases. I think the main differentiator is the traceability component, the first that you have full control on which data it will refer to, because if you deal with open source models, most of them are open, but the data it has been trained on has not been opened or given public so with vector database you introduce a step of control and explainability. Explainability means if you receive a certain answer based on your prompt, you can trace it back to the exact source where the embedding has been stored or the source of where the information is coming from and things. So this is a major use case for us for those kind of high stake solution is that you have the explainability and traceability. Explainability. It could be as simple as a semantical similarity to the text, but also the traceability of where it's coming from and the exact link of where it's coming from. So it should be, it shouldn't be referred. You can close and you can cut the line of the model referring to its previous knowledge by introducing a vector database, for example.
|
||||
Would be amazing in such specific use cases. I think the main differentiator is the traceability component, the first that you have full control on which data it will refer to, because if you deal with open source models, most of them are open, but the data it has been trained on has not been opened or given public so with vector database you introduce a step of control and explainability. Explainability means if you receive a certain answer based on your prompt, you can trace it back to the exact source where the embedding has been stored or the source of where the information is coming from and things. So this is a major use case for us for those kind of high stake solution is that you have the explainability and traceability. Explainability. It could be as simple as a semantical similarity to the text, but also the traceability of where it's coming from and the exact link of where it's coming from. So it should be, it shouldn't be referred. You can close and you can cut the line of the model referring to its previous knowledge by introducing a [vector database](https://qdrant.tech/articles/what-is-a-vector-database/), for example.
|
||||
|
||||
Iveta Lohovska:
|
||||
So there could be many other implications and improvements in terms of speed and just handling huge amounts of data, yet also nice to have that come with this kind of technique, but the prior use case is actually not incentivized around those.
|
||||
@@ -140,7 +139,7 @@ Iveta Lohovska:
|
||||
Yeah, so most of the cases, I would say 99% of the cases, is that if you have such a high requirements around security and explainability, security of the data, but those security of the whole use case and environment, and the explainability and trustworthiness of the answer, then it's very natural to have expectations that will be on prem and not in the cloud, because only on prem you have a full control of where your data sits, where your model sits, the full ownership of your IP, and then the full ownership of having less question marks of the implementation and architecture, but mainly the full ownership of the end to end solution. So when it comes to those use cases, RAG on Prem, with the whole infrastructure, with the whole software and platform layers, including models on Prem, not accessible through an API, through a service somewhere where you don't know where the guardrails is, who designed the guardrails, what are the guardrails? And we see those, this a lot with, for example, copilot, a lot of question marks around that. So it's a huge part of my work is just talking of it, just sorting out that.
|
||||
|
||||
Sabrina Aquino:
|
||||
Exactly. You don't want to just give away your data to a cloud provider, because there's many implications that that comes with. And I think even your clients, they need certain certifications, then they need to make sure that nobody can access that data, something that you cannot. Exactly. I think ensure if you're just using a cloud provider somewhere, which is, I think something that's very important when you're thinking about these high stakes solutions. But also I think if you're going to maybe outsource some of the infrastructure, you also need to think about something that's similar to a hybrid cloud solution where you can keep your data and outsource the kind of management of infrastructure. So that's also a nice use case for that, right?
|
||||
Exactly. You don't want to just give away your data to a cloud provider, because there's many implications that that comes with. And I think even your clients, they need certain certifications, then they need to make sure that nobody can access that data, something that you cannot. Exactly. I think ensure if you're just using a cloud provider somewhere, which is, I think something that's very important when you're thinking about these high stakes solutions. But also I think if you're going to maybe outsource some of the infrastructure, you also need to think about something that's similar to a [hybrid cloud solution](https://qdrant.tech/documentation/hybrid-cloud/) where you can keep your data and outsource the kind of management of infrastructure. So that's also a nice use case for that, right?
|
||||
|
||||
Iveta Lohovska:
|
||||
I mean, I work for HPE, so hybrid is like one of our biggest sacred words. Yeah, exactly. But actually like if you see the trends and if you see how expensive is to work to run some of those workloads in the cloud, either for training for national model or fine tuning. And no one talks about inference, inference not in ten users, but inference in hundred users with big organizations. This itself is not sustainable. Honestly, when you do the simple Linux, algebra or math of the exponential cost around this. That's why everything is hybrid. And there are use cases that make sense to be fast and speedy and easy to play with, low risk in the cloud to try.
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed vector database that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product.
|
||||
In their mission to support large-scale AI innovation, [Airbyte](https://airbyte.com/) and Qdrant are collaborating on the launch of Qdrant’s new offering - [Qdrant Hybrid Cloud](/hybrid-cloud/). This collaboration allows users to leverage the synergistic capabilities of both Airbyte and Qdrant within a private infrastructure. Qdrant’s new offering represents the first managed vector database that can be deployed in any environment. Businesses optimizing their data infrastructure with Airbyte are now able to host a vector database either on premise, or on a public cloud of their choice - while still reaping the benefits of a managed database product.
|
||||
|
||||
This is a major step forward in offering enterprise customers incredible synergy for maximizing the potential of their AI data. Qdrant's new Kubernetes-native design, coupled with Airbyte’s powerful data ingestion pipelines meet the needs of developers who are both prototyping and building production-level apps. Airbyte simplifies the process of data integration by providing a platform that connects to various sources and destinations effortlessly. Moreover, Qdrant Hybrid Cloud leverages advanced indexing and search capabilities to empower users to explore and analyze their data efficiently.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
[Aleph Alpha](https://aleph-alpha.com/) and Qdrant are on a joint mission to empower the world’s best companies in their AI journey. The launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) furthers this effort by ensuring complete data sovereignty and hosting security. This latest collaboration is all about giving enterprise customers complete transparency and sovereignty to make use of AI in their own environment. By using a hybrid cloud vector database, those looking to leverage vector search for the AI applications can now ensure their proprietary and customer data is completely secure.
|
||||
[Aleph Alpha](https://aleph-alpha.com/) and Qdrant are on a joint mission to empower the world’s best companies in their AI journey. The launch of [Qdrant Hybrid Cloud](/hybrid-cloud/) furthers this effort by ensuring complete data sovereignty and hosting security. This latest collaboration is all about giving enterprise customers complete transparency and sovereignty to make use of AI in their own environment. By using a hybrid cloud vector database, those looking to leverage vector search for the AI applications can now ensure their proprietary and customer data is completely secure.
|
||||
|
||||
Aleph Alpha’s state-of-the-art technology, offering unmatched quality and safety, cater perfectly to large-scale business applications and complex scenarios utilized by professionals across fields such as science, law, and security globally. Recognizing that these sophisticated use cases often demand comprehensive data processing capabilities beyond what standalone LLMs can provide, the collaboration between Aleph Alpha and Qdrant Hybrid Cloud introduces a robust platform. This platform empowers customers with full data sovereignty, enabling secure management of highly specific and sensitive information within their own infrastructure.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited to share that Qdrant and [Cohere](https://cohere.com/) are partnering on the launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) to enable global audiences to build and scale their AI applications quickly and securely. With Cohere's world-class large language models (LLMs), getting the most out of vector search becomes incredibly easy. Qdrant's new Hybrid Cloud offering and its Kubernetes-native design can be coupled with Cohere's powerful models and APIs. This combination allows for simple setup when prototyping and deploying AI solutions.
|
||||
We’re excited to share that Qdrant and [Cohere](https://cohere.com/) are partnering on the launch of [Qdrant Hybrid Cloud](/hybrid-cloud/) to enable global audiences to build and scale their AI applications quickly and securely. With Cohere's world-class large language models (LLMs), getting the most out of vector search becomes incredibly easy. Qdrant's new Hybrid Cloud offering and its Kubernetes-native design can be coupled with Cohere's powerful models and APIs. This combination allows for simple setup when prototyping and deploying AI solutions.
|
||||
|
||||
It’s no secret that Retrieval Augmented Generation (RAG) has shown to be a powerful method of building conversational AI products, such as chatbots or customer support systems. With Cohere's managed LLM service, scientists and developers can tap into state-of-the-art text generation and understanding capabilities, all accessible via API. Qdrant Hybrid Cloud seamlessly integrates with Cohere’s foundation models, enabling convenient data vectorization and highly accurate semantic search.
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ tags:
|
||||
|
||||
Developers are constantly seeking new ways to enhance their AI applications with new customer experiences. At the core of this are vector databases, as they enable the efficient handling of complex, unstructured data, making it possible to power applications with semantic search, personalized recommendation systems, and intelligent Q&A platforms. However, when deploying such new AI applications, especially those handling sensitive or personal user data, privacy becomes important.
|
||||
|
||||
[DigitalOcean](https://www.digitalocean.com/) and Qdrant are actively addressing this with an integration that lets developers deploy a managed vector database in their existing DigitalOcean environments. With the recent launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), developers can seamlessly deploy Qdrant on DigitalOcean Kubernetes (DOKS) clusters, making it easier for developers to handle vector databases without getting bogged down in the complexity of managing the underlying infrastructure.
|
||||
[DigitalOcean](https://www.digitalocean.com/) and Qdrant are actively addressing this with an integration that lets developers deploy a managed vector database in their existing DigitalOcean environments. With the recent launch of [Qdrant Hybrid Cloud](/hybrid-cloud/), developers can seamlessly deploy Qdrant on DigitalOcean Kubernetes (DOKS) clusters, making it easier for developers to handle vector databases without getting bogged down in the complexity of managing the underlying infrastructure.
|
||||
|
||||
#### Unlocking the Power of Generative AI with Qdrant and DigitalOcean
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) offering, allowing developers to deploy a managed vector database in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations.
|
||||
We’re excited to share that Qdrant and [Haystack](https://haystack.deepset.ai/) are continuing to expand their seamless integration to the new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering, allowing developers to deploy a managed vector database in their own environment of choice. Earlier this year, both Qdrant and Haystack, started to address their user’s growing need for production-ready retrieval-augmented-generation (RAG) deployments. The ability to build and deploy AI apps anywhere now allows for complete data sovereignty and control. This gives large enterprise customers the peace of mind they need before they expand AI functionalities throughout their operations.
|
||||
|
||||
With a highly customizable framework like Haystack, implementing vector search becomes incredibly simple. Qdrant's new Qdrant Hybrid Cloud offering and its Kubernetes-native design supports customers all the way from a simple prototype setup to a production scenario on any hosting platform. Users can attach AI functionalities to their existing in-house software by creating custom integration components. Don’t forget, both products are open-source and highly modular!
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We're thrilled to announce the collaboration between Qdrant and [Jina AI](https://jina.ai/) for the launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), empowering users worldwide to rapidly and securely develop and scale their AI applications. By leveraging Jina AI's top-tier large language models (LLMs), engineers and scientists can optimize their vector search efforts. Qdrant's latest Hybrid Cloud solution, designed natively with Kubernetes, seamlessly integrates with Jina AI's robust embedding models and APIs. This synergy streamlines both prototyping and deployment processes for AI solutions.
|
||||
We're thrilled to announce the collaboration between Qdrant and [Jina AI](https://jina.ai/) for the launch of [Qdrant Hybrid Cloud](/hybrid-cloud/), empowering users worldwide to rapidly and securely develop and scale their AI applications. By leveraging Jina AI's top-tier large language models (LLMs), engineers and scientists can optimize their vector search efforts. Qdrant's latest Hybrid Cloud solution, designed natively with Kubernetes, seamlessly integrates with Jina AI's robust embedding models and APIs. This synergy streamlines both prototyping and deployment processes for AI solutions.
|
||||
|
||||
Retrieval Augmented Generation (RAG) is broadly adopted as the go-to Generative AI solution, as it enables powerful and cost-effective chatbots, customer support agents and other forms of semantic search applications. Through Jina AI's managed service, users gain access to cutting-edge text generation and comprehension capabilities, conveniently accessible through an API. Qdrant Hybrid Cloud effortlessly incorporates Jina AI's embedding models, facilitating smooth data vectorization and delivering exceptionally precise semantic search functionality.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
[LangChain](https://www.langchain.com/) and Qdrant are collaborating on the launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), which is designed to empower engineers and scientists globally to easily and securely develop and scale their GenAI applications. Harnessing LangChain’s robust framework, users can unlock the full potential of vector search, enabling the creation of stable and effective AI products. Qdrant Hybrid Cloud extends the same powerful functionality of Qdrant onto a Kubernetes-based architecture, enhancing LangChain’s capability to cater to users across any environment.
|
||||
[LangChain](https://www.langchain.com/) and Qdrant are collaborating on the launch of [Qdrant Hybrid Cloud](/hybrid-cloud/), which is designed to empower engineers and scientists globally to easily and securely develop and scale their GenAI applications. Harnessing LangChain’s robust framework, users can unlock the full potential of vector search, enabling the creation of stable and effective AI products. Qdrant Hybrid Cloud extends the same powerful functionality of Qdrant onto a Kubernetes-based architecture, enhancing LangChain’s capability to cater to users across any environment.
|
||||
|
||||
Qdrant Hybrid Cloud provides users with the flexibility to deploy their vector database in a preferred environment. Through container-based scalable deployments, companies can leverage cutting-edge frameworks like LangChain while maintaining compatibility with their existing hosting architecture for data sources, embedded models, and LLMs. This potent combination empowers organizations to develop robust and secure applications capable of text-based search, complex question-answering, recommendations and analysis.
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ tags:
|
||||
- launch partners
|
||||
---
|
||||
|
||||
With the launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) we provide developers the ability to deploy Qdrant as a managed vector database in any desired environment, be it *in the cloud, on premise, or on the edge*.
|
||||
With the launch of [Qdrant Hybrid Cloud](/hybrid-cloud/) we provide developers the ability to deploy Qdrant as a managed vector database in any desired environment, be it *in the cloud, on premise, or on the edge*.
|
||||
|
||||
We are excited to have trusted industry players support the launch of Qdrant Hybrid Cloud, allowing developers to unlock best-in-class advantages for building production-ready AI applications:
|
||||
|
||||
@@ -97,8 +97,8 @@ Additionally, we built comprehensive documentation tutorials on how to successfu
|
||||
|
||||
#### Get Started Now!
|
||||
|
||||
[Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) marks a significant advancement in vector databases, offering the most flexible way to implement vector search.
|
||||
[Qdrant Hybrid Cloud](/hybrid-cloud/) marks a significant advancement in vector databases, offering the most flexible way to implement vector search.
|
||||
|
||||
You can test out Qdrant Hybrid Cloud today! Simply sign up for or log into your [Qdrant Cloud account](https://cloud.qdrant.io/login) and get started in the **Hybrid Cloud** section. Also, to learn more about Qdrant Hybrid Cloud read our [Official Release Blog](/blog/hybrid-cloud/) or our [Qdrant Hybrid Cloud website](https://hybrid-cloud.qdrant.tech/). For additional technical insights, please read our [documentation](/documentation/hybrid-cloud/).
|
||||
You can test out Qdrant Hybrid Cloud today! Simply sign up for or log into your [Qdrant Cloud account](https://cloud.qdrant.io/login) and get started in the **Hybrid Cloud** section. Also, to learn more about Qdrant Hybrid Cloud read our [Official Release Blog](/blog/hybrid-cloud/) or our [Qdrant Hybrid Cloud website](/hybrid-cloud/). For additional technical insights, please read our [documentation](/documentation/hybrid-cloud/).
|
||||
|
||||
[](https://cloud.qdrant.io/login)
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We're happy to announce the collaboration between [LlamaIndex](https://www.llamaindex.ai/) and [Qdrant’s new Hybrid Cloud launch](https://hybrid-cloud.qdrant.tech/), aimed at empowering engineers and scientists worldwide to swiftly and securely develop and scale their GenAI applications. By leveraging LlamaIndex's robust framework, users can maximize the potential of vector search and create stable and effective AI products. Qdrant Hybrid Cloud offers the same Qdrant functionality on a Kubernetes-based architecture, which further expands the ability of LlamaIndex to support any user on any environment.
|
||||
We're happy to announce the collaboration between [LlamaIndex](https://www.llamaindex.ai/) and [Qdrant’s new Hybrid Cloud launch](/hybrid-cloud/), aimed at empowering engineers and scientists worldwide to swiftly and securely develop and scale their GenAI applications. By leveraging LlamaIndex's robust framework, users can maximize the potential of vector search and create stable and effective AI products. Qdrant Hybrid Cloud offers the same Qdrant functionality on a Kubernetes-based architecture, which further expands the ability of LlamaIndex to support any user on any environment.
|
||||
|
||||
With Qdrant Hybrid Cloud, users have the flexibility to deploy their vector database in an environment of their choice. By using container-based scalable deployments, companies can leverage a cutting-edge framework like LlamaIndex, while staying deployed in the same hosting architecture as data sources, embedding models and LLMs. This powerful combination empowers organizations to build strong and secure applications that search, understand meaning and converse in text.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
Qdrant and [Oracle Cloud Infrastructure (OCI) Cloud Engineering](https://www.oracle.com/cloud/) are thrilled to announce the ability to deploy [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) as a managed service on OCI. This marks the next step in the collaboration between Qdrant and Oracle Cloud Infrastructure, which will enable enterprises to realize the benefits of artificial intelligence powered through scalable vector search. In 2023, OCI added Qdrant to its [Oracle Cloud Infrastructure solution portfolio](https://blogs.oracle.com/cloud-infrastructure/post/vecto-database-qdrant-support-oci-kubernetes). Qdrant Hybrid Cloud is the managed service of the Qdrant vector search engine that can be deployed and run in any existing OCI environment, allowing enterprises to run fully managed vector search workloads in their existing infrastructure. This is a milestone for leveraging a managed vector search engine for data-sensitive AI applications.
|
||||
Qdrant and [Oracle Cloud Infrastructure (OCI) Cloud Engineering](https://www.oracle.com/cloud/) are thrilled to announce the ability to deploy [Qdrant Hybrid Cloud](/hybrid-cloud/) as a managed service on OCI. This marks the next step in the collaboration between Qdrant and Oracle Cloud Infrastructure, which will enable enterprises to realize the benefits of artificial intelligence powered through scalable vector search. In 2023, OCI added Qdrant to its [Oracle Cloud Infrastructure solution portfolio](https://blogs.oracle.com/cloud-infrastructure/post/vecto-database-qdrant-support-oci-kubernetes). Qdrant Hybrid Cloud is the managed service of the Qdrant vector search engine that can be deployed and run in any existing OCI environment, allowing enterprises to run fully managed vector search workloads in their existing infrastructure. This is a milestone for leveraging a managed vector search engine for data-sensitive AI applications.
|
||||
|
||||
In the past years, enterprises have been actively engaged in exploring AI applications to enhance their products and services or unlock internal company knowledge to drive the productivity of teams. These applications range from generative AI use cases, for example, powered by retrieval augmented generation (RAG), recommendation systems, or advanced enterprise search through semantic, similarity, or neural search. As these vector search applications continue to evolve and grow with respect to dimensionality and complexity, it will be increasingly relevant to have a scalable, manageable vector search engine, also called out by Gartner’s 2024 Impact Radar. In addition to scalability, enterprises also require flexibility in deployment options to be able to maximize the use of these new AI tools within their existing environment, ensuring interoperability and full control over their data.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
With the official release of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), businesses running their data infrastructure on [OVHcloud](https://ovhcloud.com/) are now able to deploy a fully managed vector database in their existing OVHcloud environment. We are excited about this partnership, which has been established through the [OVHcloud Open Trusted Cloud](https://opentrustedcloud.ovhcloud.com/en/) program, as it is based on our shared understanding of the importance of trust, control, and data privacy in the context of the emerging landscape of enterprise-grade AI applications. As part of this collaboration, we are also providing a detailed use case tutorial on building a recommendation system that demonstrates the benefits of running Qdrant Hybrid Cloud on OVHcloud.
|
||||
With the official release of [Qdrant Hybrid Cloud](/hybrid-cloud/), businesses running their data infrastructure on [OVHcloud](https://ovhcloud.com/) are now able to deploy a fully managed vector database in their existing OVHcloud environment. We are excited about this partnership, which has been established through the [OVHcloud Open Trusted Cloud](https://opentrustedcloud.ovhcloud.com/en/) program, as it is based on our shared understanding of the importance of trust, control, and data privacy in the context of the emerging landscape of enterprise-grade AI applications. As part of this collaboration, we are also providing a detailed use case tutorial on building a recommendation system that demonstrates the benefits of running Qdrant Hybrid Cloud on OVHcloud.
|
||||
|
||||
Deploying Qdrant Hybrid Cloud on OVHcloud's infrastructure represents a significant leap for European businesses invested in AI-driven projects, as this collaboration underscores the commitment to meeting the rigorous requirements for data privacy and control of European startups and enterprises building AI solutions. As businesses are progressing on their AI journey, they require dedicated solutions that allow them to make their data accessible for machine learning and AI projects, without having it leave the company's security perimeter. Prioritizing data sovereignty, a crucial aspect in today's digital landscape, will help startups and enterprises accelerate their AI agendas and build even more differentiating AI-enabled applications. The ability of running Qdrant Hybrid Cloud on OVHcloud not only underscores the commitment to innovative, secure AI solutions but also ensures that companies can navigate the complexities of AI and machine learning workloads with the flexibility and security required.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited about our collaboration with Red Hat to bring the Qdrant vector database to [Red Hat OpenShift](https://www.redhat.com/en/technologies/cloud-computing/openshift) customers! With the release of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), developers can now deploy and run the Qdrant vector database directly in their Red Hat OpenShift environment. This collaboration enables developers to scale more seamlessly, operate more consistently across hybrid cloud environments, and maintain complete control over their vector data. This is a big step forward in simplifying AI infrastructure and empowering data-driven projects, like retrieval augmented generation (RAG) use cases, advanced search scenarios, or recommendations systems.
|
||||
We’re excited about our collaboration with Red Hat to bring the Qdrant vector database to [Red Hat OpenShift](https://www.redhat.com/en/technologies/cloud-computing/openshift) customers! With the release of [Qdrant Hybrid Cloud](/hybrid-cloud/), developers can now deploy and run the Qdrant vector database directly in their Red Hat OpenShift environment. This collaboration enables developers to scale more seamlessly, operate more consistently across hybrid cloud environments, and maintain complete control over their vector data. This is a big step forward in simplifying AI infrastructure and empowering data-driven projects, like retrieval augmented generation (RAG) use cases, advanced search scenarios, or recommendations systems.
|
||||
|
||||
In the rapidly evolving field of Artificial Intelligence and Machine Learning, the demand for being able to manage the modern AI stack within the existing infrastructure becomes increasingly relevant for businesses. As enterprises are launching new AI applications and use cases into production, they require the ability to maintain complete control over their data, since these new apps often work with sensitive internal and customer-centric data that needs to remain within the owned premises. This is why enterprises are increasingly looking for maximum deployment flexibility for their AI workloads.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
In a move to empower the next wave of AI innovation, Qdrant and [Scaleway](https://www.scaleway.com/en/) collaborate to introduce [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), a fully managed vector database that can be deployed on existing Scaleway environments. This collaboration is set to democratize access to advanced AI capabilities, enabling developers to easily deploy and scale vector search technologies within Scaleway's robust and developer-friendly cloud infrastructure. By focusing on the unique needs of startups and the developer community, Qdrant and Scaleway are providing access to intuitive and easy to use tools, making cutting-edge AI more accessible than ever before.
|
||||
In a move to empower the next wave of AI innovation, Qdrant and [Scaleway](https://www.scaleway.com/en/) collaborate to introduce [Qdrant Hybrid Cloud](/hybrid-cloud/), a fully managed vector database that can be deployed on existing Scaleway environments. This collaboration is set to democratize access to advanced AI capabilities, enabling developers to easily deploy and scale vector search technologies within Scaleway's robust and developer-friendly cloud infrastructure. By focusing on the unique needs of startups and the developer community, Qdrant and Scaleway are providing access to intuitive and easy to use tools, making cutting-edge AI more accessible than ever before.
|
||||
|
||||
Building on this vision, the integration between Scaleway and Qdrant Hybrid Cloud leverages the strengths of both Qdrant, with its leading open-source vector database, and Scaleway, known for its innovative and scalable cloud solutions. This integration means startups and developers can now harness the power of vector search - essential for AI applications like recommendation systems, image recognition, and natural language processing - within their existing environment without the complexity of maintaining such advanced setups.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
Qdrant and [STACKIT](https://www.stackit.de/en/) are thrilled to announce that developers are now able to deploy a fully managed vector database to their STACKIT environment with the introduction of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/). This is a great step forward for the German AI ecosystem as it enables developers and businesses to build cutting edge AI applications that run on German data centers with full control over their data.
|
||||
Qdrant and [STACKIT](https://www.stackit.de/en/) are thrilled to announce that developers are now able to deploy a fully managed vector database to their STACKIT environment with the introduction of [Qdrant Hybrid Cloud](/hybrid-cloud/). This is a great step forward for the German AI ecosystem as it enables developers and businesses to build cutting edge AI applications that run on German data centers with full control over their data.
|
||||
|
||||
Vector databases are an essential component of the modern AI stack. They enable rapid and accurate retrieval of high-dimensional data, crucial for powering search, recommendation systems, and augmenting machine learning models. In the rising field of GenAI, vector databases power retrieval-augmented-generation (RAG) scenarios as they are able to enhance the output of large language models (LLMs) by injecting relevant contextual information. However, this contextual information is often rooted in confidential internal or customer-related information, which is why enterprises are in pursuit of solutions that allow them to make this data available for their AI applications without compromising data privacy, losing data control, or letting data exit the company's secure environment.
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
We’re excited to share that Qdrant and [Vultr](https://www.vultr.com/) are partnering to provide seamless scalability and performance for vector search workloads. With Vultr's global footprint and customizable platform, deploying vector search workloads becomes incredibly flexible. Qdrant's new [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) offering and its Kubernetes-native design, coupled with Vultr's straightforward virtual machine provisioning, allows for simple setup when prototyping and building next-gen AI apps.
|
||||
We’re excited to share that Qdrant and [Vultr](https://www.vultr.com/) are partnering to provide seamless scalability and performance for vector search workloads. With Vultr's global footprint and customizable platform, deploying vector search workloads becomes incredibly flexible. Qdrant's new [Qdrant Hybrid Cloud](/hybrid-cloud/) offering and its Kubernetes-native design, coupled with Vultr's straightforward virtual machine provisioning, allows for simple setup when prototyping and building next-gen AI apps.
|
||||
|
||||
#### Adapting to Diverse AI Development Needs with Customization and Deployment Flexibility
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ tags:
|
||||
- Hybrid Cloud
|
||||
---
|
||||
|
||||
We are excited to announce the official launch of [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) today, a significant leap forward in the field of vector search and enterprise AI. Rooted in our open-source origin, we are committed to offering our users and customers unparalleled control and sovereignty over their data and vector search workloads. Qdrant Hybrid Cloud stands as **the industry's first managed vector database that can be deployed in any environment** - be it cloud, on-premise, or the edge.
|
||||
We are excited to announce the official launch of [Qdrant Hybrid Cloud](/hybrid-cloud/) today, a significant leap forward in the field of vector search and enterprise AI. Rooted in our open-source origin, we are committed to offering our users and customers unparalleled control and sovereignty over their data and vector search workloads. Qdrant Hybrid Cloud stands as **the industry's first managed vector database that can be deployed in any environment** - be it cloud, on-premise, or the edge.
|
||||
|
||||
<p align="center"><iframe width="560" height="315" src="https://www.youtube.com/embed/gWH2uhWgTvM" title="YouTube video player" frameborder="0" allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture; web-share" allowfullscreen></iframe></p>
|
||||
|
||||
|
||||
@@ -1,14 +1,11 @@
|
||||
---
|
||||
draft: false
|
||||
title: Open Source Vector Search Engine and Vector Database - Andrey Vasnetsov
|
||||
title: Optimizing an Open Source Vector Database with Andrey Vasnetsov
|
||||
slug: open-source-vector-search-engine-vector-database
|
||||
short_description: CTO of Qdrant Andrey talks about Vector search engines and
|
||||
the technical facets and challenges encountered in developing an open-source
|
||||
vector database.
|
||||
description: Andrey Vasnetsov, CTO and Co-founder of Qdrant, presents an
|
||||
in-depth look into the intricacies of their open-source vector search engine
|
||||
and database, detailing its optimized architecture, data structure challenges,
|
||||
and innovative filtering techniques for efficient vector similarity searches.
|
||||
description: Learn key strategies for optimizing vector search from Andrey Vasnetsov, CTO at Qdrant. Dive into techniques like efficient indexing for improved performance.
|
||||
preview_image: /blog/from_cms/andrey-vasnetsov-cropped.png
|
||||
date: 2024-01-10T16:04:57.804Z
|
||||
author: Demetrios Brinkmann
|
||||
@@ -18,13 +15,16 @@ tags:
|
||||
- Vector Search Engine
|
||||
- Vector Database
|
||||
---
|
||||
|
||||
# Optimizing Open Source Vector Search: Strategies from Andrey Vasnetsov at Qdrant
|
||||
|
||||
> *"For systems like Qdrant, scalability and performance in my opinion, is much more important than transactional consistency, so it should be treated as a search engine rather than database."*\
|
||||
-- Andrey Vasnetsov
|
||||
>
|
||||
|
||||
Discussing core differences between search engines and databases, Andrey underlined the importance of application needs and scalability in database selection for vector search tasks.
|
||||
|
||||
Andrey Vasnetsov, CTO at Qdrant is an enthusiast of Open Source, machine learning, and vector search. He works on Open Source projects related to Vector Similarity Search and Similarity Learning. He prefers practical over theoretical, working demo over arXiv paper.
|
||||
Andrey Vasnetsov, CTO at Qdrant is an enthusiast of [Open Source](https://qdrant.tech/), machine learning, and vector search. He works on Open Source projects related to [Vector Similarity Search](https://qdrant.tech/articles/vector-similarity-beyond-search/) and Similarity Learning. He prefers practical over theoretical, working demo over arXiv paper.
|
||||
|
||||
***You can watch this episode on [YouTube](https://www.youtube.com/watch?v=bU38Ovdh3NY).***
|
||||
|
||||
@@ -34,7 +34,7 @@ Andrey Vasnetsov, CTO at Qdrant is an enthusiast of Open Source, machine learnin
|
||||
|
||||
## **Top Takeaways:**
|
||||
|
||||
Dive into the intricacies of vector databases with Andrey as he unpacks Qdrant's approach to combining filtering and vector search, revealing how in-place filtering during graph traversal optimizes precision without sacrificing search exactness, even when scaling to billions of vectors.
|
||||
Dive into the intricacies of [vector databases](https://qdrant.tech/articles/what-is-a-vector-database/) with Andrey as he unpacks Qdrant's approach to combining filtering and vector search, revealing how in-place filtering during graph traversal optimizes precision without sacrificing search exactness, even when scaling to billions of vectors.
|
||||
|
||||
5 key insights you’ll learn:
|
||||
|
||||
@@ -48,7 +48,7 @@ Dive into the intricacies of vector databases with Andrey as he unpacks Qdrant's
|
||||
|
||||
- 🔗 **Connected Graph Challenges:** Learn about navigating the difficulties of maintaining a connected graph while filtering during search operations.
|
||||
|
||||
> Fun Fact: The Qdrant system is capable of in-place filtering during graph traversal, which is a novel approach compared to traditional post-filtering methods, ensuring the correct quantity of results that meet the filtering conditions.
|
||||
> Fun Fact: [The Qdrant system](https://qdrant.tech/) is capable of in-place filtering during graph traversal, which is a novel approach compared to traditional post-filtering methods, ensuring the correct quantity of results that meet the filtering conditions.
|
||||
>
|
||||
|
||||
## Timestamps:
|
||||
|
||||
@@ -0,0 +1,574 @@
|
||||
---
|
||||
title: "Qdrant 1.10 - Universal Query, Built-in IDF & ColBERT Support"
|
||||
draft: false
|
||||
short_description: "Single search API. Server-side IDF. Native multivector support."
|
||||
description: "Consolidated search API, built-in IDF, and native multivector support. "
|
||||
preview_image: /blog/qdrant-1.10.x/social_preview.png
|
||||
social_preview_image: /blog/qdrant-1.10.x/social_preview.png
|
||||
date: 2024-07-01T00:00:00-08:00
|
||||
author: David Myriel
|
||||
featured: false
|
||||
tags:
|
||||
- vector search
|
||||
- ColBERT late interaction
|
||||
- BM25 algorithm
|
||||
- search API
|
||||
- new features
|
||||
---
|
||||
|
||||
[Qdrant 1.10.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.10.0) This version introduces some major changes, so let's dive right in:
|
||||
|
||||
**Universal Query API:** All search APIs, including Hybrid Search, are now in one Query endpoint.</br>
|
||||
**Built-in IDF:** We added the IDF mechanism to Qdrant's core search and indexing processes.</br>
|
||||
**Multivector Support:** Native support for late interaction ColBERT is accessible via Query API.
|
||||
|
||||
## One Endpoint for All Queries
|
||||
**Query API** will consolidate all search APIs into a single request. Previously, you had to work outside of the API to combine different search requests. Now these approaches are reduced to parameters of a single request, so you can avoid merging individual results.
|
||||
|
||||
You can now configure the Query API request with the following parameters:
|
||||
|
||||
|Parameter|Description|
|
||||
|-|-|
|
||||
|no parameter|Returns points by `id`|
|
||||
|`nearest`|Queries nearest neighbors ([Search](/documentation/concepts/search/))|
|
||||
|`fusion`|Fuses sparse/dense prefetch queries ([Hybrid Search](/documentation/concepts/hybrid-queries/#hybrid-search))|
|
||||
|`discover`|Queries `target` with added `context` ([Discovery](/documentation/concepts/explore/#discovery-api))|
|
||||
|`context` |No target with `context` only ([Context](/documentation/concepts/explore/#context-search))|
|
||||
|`recommend`|Queries against `positive`/`negative` examples. ([Recommendation](/documentation/concepts/explore/#recommendation-api))|
|
||||
|`order_by`|Orders results by [payload field](/documentation/concepts/hybrid-queries/#re-ranking-with-payload-values)|
|
||||
|
||||
For example, you can configure Query API to run [Discovery search](/documentation/concepts/explore/#discovery-api). Let's see how that looks:
|
||||
|
||||
```http
|
||||
POST collections/{collection_name}/points/query
|
||||
{
|
||||
"query": {
|
||||
"discover": {
|
||||
"target": <vector_input>,
|
||||
"context": [
|
||||
{
|
||||
"positive": <vector_input>,
|
||||
"negative": <vector_input>
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
We will be publishing code samples in [docs](/documentation/concepts/hybrid-queries/) and our new [API specification](http://api.qdrant.tech).</br> *If you need additional support with this new method, our [Discord](https://qdrant.to/discord) on-call engineers can help you.*
|
||||
|
||||
### Native Hybrid Search Support
|
||||
Query API now also natively supports **sparse/dense fusion**. Up to this point, you had to combine the results of sparse and dense searches on your own. This is now sorted on the back-end, and you only have to configure them as basic parameters for Query API.
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
{
|
||||
"prefetch": [
|
||||
{
|
||||
"query": {
|
||||
"indices": [1, 42], // <┐
|
||||
"values": [0.22, 0.8] // <┴─sparse vector
|
||||
},
|
||||
"using": "sparse",
|
||||
"limit": 20
|
||||
},
|
||||
{
|
||||
"query": [0.01, 0.45, 0.67, ...], // <-- dense vector
|
||||
"using": "dense",
|
||||
"limit": 20
|
||||
}
|
||||
],
|
||||
"query": { "fusion": "rrf" }, // <--- reciprocal rank fusion
|
||||
"limit": 10
|
||||
}
|
||||
```
|
||||
|
||||
```rust
|
||||
use qdrant_client::Qdrant;
|
||||
use qdrant_client::qdrant::{Fusion, PrefetchQueryBuilder, Query, QueryPointsBuilder};
|
||||
|
||||
let client = Qdrant::from_url("http://localhost:6334").build()?;
|
||||
|
||||
client.query(
|
||||
QueryPointsBuilder::new("{collection_name}")
|
||||
.add_prefetch(PrefetchQueryBuilder::default()
|
||||
.query(Query::new_nearest([(1, 0.22), (42, 0.8)].as_slice()))
|
||||
.using("sparse")
|
||||
.limit(20u64)
|
||||
)
|
||||
.add_prefetch(PrefetchQueryBuilder::default()
|
||||
.query(Query::new_nearest(vec![0.01, 0.45, 0.67]))
|
||||
.using("dense")
|
||||
.limit(20u64)
|
||||
)
|
||||
.query(Query::new_fusion(Fusion::Rrf))
|
||||
).await?;
|
||||
```
|
||||
|
||||
```java
|
||||
import static io.qdrant.client.QueryFactory.nearest;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import static io.qdrant.client.QueryFactory.fusion;
|
||||
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.grpc.Points.Fusion;
|
||||
import io.qdrant.client.grpc.Points.PrefetchQuery;
|
||||
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||
|
||||
QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
|
||||
|
||||
client.queryAsync(
|
||||
QueryPoints.newBuilder()
|
||||
.setCollectionName("{collection_name}")
|
||||
.addPrefetch(PrefetchQuery.newBuilder()
|
||||
.setQuery(nearest(List.of(0.22f, 0.8f), List.of(1, 42)))
|
||||
.setUsing("sparse")
|
||||
.setLimit(20)
|
||||
.build())
|
||||
.addPrefetch(PrefetchQuery.newBuilder()
|
||||
.setQuery(nearest(List.of(0.01f, 0.45f, 0.67f)))
|
||||
.setUsing("dense")
|
||||
.setLimit(20)
|
||||
.build())
|
||||
.setQuery(fusion(Fusion.RRF))
|
||||
.build())
|
||||
.get();
|
||||
```
|
||||
|
||||
```csharp
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
|
||||
var client = new QdrantClient("localhost", 6334);
|
||||
|
||||
await client.QueryAsync(
|
||||
collectionName: "{collection_name}",
|
||||
prefetch: new List < PrefetchQuery > {
|
||||
new() {
|
||||
Query = new(float, uint)[] {
|
||||
(0.22f, 1), (0.8f, 42),
|
||||
},
|
||||
Using = "sparse",
|
||||
Limit = 20
|
||||
},
|
||||
new() {
|
||||
Query = new float[] {
|
||||
0.01f, 0.45f, 0.67f
|
||||
},
|
||||
Using = "dense",
|
||||
Limit = 20
|
||||
}
|
||||
},
|
||||
query: Fusion.Rrf
|
||||
);
|
||||
```
|
||||
|
||||
Query API can now pre-fetch vectors for requests, which means you can run queries sequentially within the same API call. There are a lot of options here, so you will need to define a strategy to merge these requests using new parameters. For example, you can now include **rescoring within Hybrid Search**, which can open the door to strategies like iterative refinement via matryoshka embeddings.
|
||||
|
||||
*To learn more about this, read the [Query API documentation](/documentation/concepts/search/#query-api).*
|
||||
|
||||
## Inverse Document Frequency [IDF]
|
||||
|
||||
IDF is a critical component of the **TF-IDF (Term Frequency-Inverse Document Frequency)** weighting scheme used to evaluate the importance of a word in a document relative to a collection of documents (corpus).
|
||||
There are various ways in which IDF might be calculated, but the most commonly used formula is:
|
||||
|
||||
$$
|
||||
\text{IDF}(q_i) = \ln \left(\frac{N - n(q_i) + 0.5}{n(q_i) + 0.5}+1\right)
|
||||
$$
|
||||
|
||||
Where:</br>
|
||||
`N` is the total number of documents in the collection. </br>
|
||||
`n` is the number of documents containing non-zero values for the given vector.
|
||||
|
||||
This variant is also used in BM25, whose support was heavily requested by our users. We decided to move the IDF calculation into the Qdrant engine itself. This type of separation allows streaming updates of the sparse embeddings while keeping the IDF calculation up-to-date.
|
||||
|
||||
The values of IDF previously had to be calculated using all the documents on the client side. However, now that Qdrant does it out of the box, you won't need to implement it anywhere else and recompute the value if some documents are removed or newly added.
|
||||
|
||||
You can enable the IDF modifier in the collection configuration:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"sparse_vectors": {
|
||||
"text": {
|
||||
"modifier": "idf"
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```python
|
||||
from qdrant_client import QdrantClient, models
|
||||
client = QdrantClient(url="http://localhost:6333")
|
||||
client.create_collection(
|
||||
collection_name="{collection_name}",
|
||||
sparse_vectors={
|
||||
"text": models.SparseVectorParams(
|
||||
modifier=models.Modifier.IDF,
|
||||
),
|
||||
},
|
||||
)
|
||||
```
|
||||
|
||||
```rust
|
||||
use qdrant_client::Qdrant;
|
||||
use qdrant_client::qdrant::{CreateCollectionBuilder, sparse_vectors_config::SparseVectorsConfigBuilder, Modifier, SparseVectorParamsBuilder};
|
||||
|
||||
let client = Qdrant::from_url("http://localhost:6334").build()?;
|
||||
|
||||
let mut config = SparseVectorsConfigBuilder::default();
|
||||
config.add_named_vector_params(
|
||||
"text",
|
||||
SparseVectorParamsBuilder::default().modifier(Modifier::Idf),
|
||||
);
|
||||
|
||||
client
|
||||
.create_collection(
|
||||
CreateCollectionBuilder::new("{collection_name}")
|
||||
.sparse_vectors_config(config),
|
||||
)
|
||||
.await?;
|
||||
```
|
||||
|
||||
```java
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||
import io.qdrant.client.grpc.Collections.Modifier;
|
||||
import io.qdrant.client.grpc.Collections.SparseVectorConfig;
|
||||
import io.qdrant.client.grpc.Collections.SparseVectorParams;
|
||||
|
||||
QdrantClient client =
|
||||
new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
|
||||
|
||||
client
|
||||
.createCollectionAsync(
|
||||
CreateCollection.newBuilder()
|
||||
.setCollectionName("{collection_name}")
|
||||
.setSparseVectorsConfig(
|
||||
SparseVectorConfig.newBuilder()
|
||||
.putMap("text", SparseVectorParams.newBuilder().setModifier(Modifier.Idf).build()))
|
||||
.build())
|
||||
.get();
|
||||
```
|
||||
|
||||
```csharp
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
|
||||
var client = new QdrantClient("localhost", 6334);
|
||||
|
||||
await client.CreateCollectionAsync(
|
||||
collectionName: "{collection_name}",
|
||||
sparseVectorsConfig: ("text", new SparseVectorParams {
|
||||
Modifier = Modifier.Idf,
|
||||
})
|
||||
);
|
||||
```
|
||||
|
||||
### IDF as Part of BM42
|
||||
|
||||
This quarter, Qdrant also introduced BM42, a novel algorithm that combines the IDF element of BM25 with transformer-based attention matrices to improve text retrieval. It utilizes attention matrices from your embedding model to determine the importance of each token in the document based on the attention value it receives.
|
||||
|
||||
We've prepared the standard `all-MiniLM-L6-v2` Sentence Transformer so [it outputs the attention values](https://huggingface.co/Qdrant/all_miniLM_L6_v2_with_attentions). Still, you can use virtually any model of your choice, as long as you have access to its parameters. This is just another reason to stick with open source technologies over proprietary systems.
|
||||
|
||||
In practical terms, the BM42 method addresses the tokenization issues and computational costs associated with SPLADE. The model is both efficient and effective across different document types and lengths, offering enhanced search performance by leveraging the strengths of both BM25 and modern transformer techniques.
|
||||
|
||||
> To learn more about IDF and BM42, read our [dedicated technical article](/articles/bm42/).
|
||||
|
||||
**You can expect BM42 to excel in scalable RAG-based scenarios where short texts are more common.** Document inference speed is much higher with BM42, which is critical for large-scale applications such as search engines, recommendation systems, and real-time decision-making systems.
|
||||
|
||||
## Multivector Support
|
||||
We are adding native support for multivector search that is compatible, e.g., with the late-interaction [ColBERT](https://github.com/stanford-futuredata/ColBERT) model. If you are working with high-dimensional similarity searches, **ColBERT is highly recommended as a reranking step in the Universal Query search.** You will experience better quality vector retrieval since ColBERT’s approach allows for deeper semantic understanding.
|
||||
|
||||
This model retains contextual information during query-document interaction, leading to better relevance scoring. In terms of efficiency and scalability benefits, documents and queries will be encoded separately, which gives an opportunity for pre-computation and storage of document embeddings for faster retrieval.
|
||||
|
||||
**Note:** *This feature supports all the original quantization compression methods, just the same as the regular search method.*
|
||||
|
||||
**Run a query with ColBERT vectors:**
|
||||
|
||||
Query API can handle exceedingly complex requests. The following example prefetches 1000 entries most similar to the given query using the `mrl_byte` named vector, then reranks them to get the best 100 matches with `full` named vector and eventually reranks them again to extract the top 10 results with the named vector called `colbert`. A single API call can now implement complex reranking schemes.
|
||||
|
||||
```http
|
||||
POST /collections/{collection_name}/points/query
|
||||
{
|
||||
"prefetch": {
|
||||
"prefetch": {
|
||||
"query": [1, 23, 45, 67], // <------ small byte vector
|
||||
"using": "mrl_byte",
|
||||
"limit": 1000
|
||||
},
|
||||
"query": [0.01, 0.45, 0.67, ...], // <-- full dense vector
|
||||
"using": "full",
|
||||
"limit": 100
|
||||
},
|
||||
"query": [ // <─┐
|
||||
[0.1, 0.2, ...], // < │
|
||||
[0.2, 0.1, ...], // < ├─ multi-vector
|
||||
[0.8, 0.9, ...] // < │
|
||||
], // <─┘
|
||||
"using": "colbert",
|
||||
"limit": 10
|
||||
}
|
||||
```
|
||||
|
||||
```rust
|
||||
use qdrant_client::Qdrant;
|
||||
use qdrant_client::qdrant::{PrefetchQueryBuilder, Query, QueryPointsBuilder};
|
||||
|
||||
let client = Qdrant::from_url("http://localhost:6334").build()?;
|
||||
|
||||
client.query(
|
||||
QueryPointsBuilder::new("{collection_name}")
|
||||
.add_prefetch(PrefetchQueryBuilder::default()
|
||||
.add_prefetch(PrefetchQueryBuilder::default()
|
||||
.query(Query::new_nearest(vec![1.0, 23.0, 45.0, 67.0]))
|
||||
.using("mlr_byte")
|
||||
.limit(1000u64)
|
||||
)
|
||||
.query(Query::new_nearest(vec![0.01, 0.45, 0.67]))
|
||||
.using("full")
|
||||
.limit(100u64)
|
||||
)
|
||||
.query(Query::new_nearest(vec![
|
||||
vec![0.1, 0.2],
|
||||
vec![0.2, 0.1],
|
||||
vec![0.8, 0.9],
|
||||
]))
|
||||
.using("colbert")
|
||||
.limit(10u64)
|
||||
).await?;
|
||||
```
|
||||
|
||||
```java
|
||||
import static io.qdrant.client.QueryFactory.nearest;
|
||||
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.grpc.Points.PrefetchQuery;
|
||||
import io.qdrant.client.grpc.Points.QueryPoints;
|
||||
|
||||
QdrantClient client =
|
||||
new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
|
||||
|
||||
client
|
||||
.queryAsync(
|
||||
QueryPoints.newBuilder()
|
||||
.setCollectionName("{collection_name}")
|
||||
.addPrefetch(
|
||||
PrefetchQuery.newBuilder()
|
||||
.addPrefetch(
|
||||
PrefetchQuery.newBuilder()
|
||||
.setQuery(nearest(1, 23, 45, 67)) // <------------- small byte vector
|
||||
.setUsing("mrl_byte")
|
||||
.setLimit(1000)
|
||||
.build())
|
||||
.setQuery(nearest(0.01f, 0.45f, 0.67f)) // <-- dense vector
|
||||
.setUsing("full")
|
||||
.setLimit(100)
|
||||
.build())
|
||||
.setQuery(
|
||||
nearest(
|
||||
new float[][] {
|
||||
{0.1f, 0.2f}, // <─┐
|
||||
{0.2f, 0.1f}, // < ├─ multi-vector
|
||||
{0.8f, 0.9f} // < ┘
|
||||
}))
|
||||
.setUsing("colbert")
|
||||
.setLimit(10)
|
||||
.build())
|
||||
.get();
|
||||
```
|
||||
|
||||
```csharp
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
|
||||
var client = new QdrantClient("localhost", 6334);
|
||||
|
||||
await client.QueryAsync(
|
||||
collectionName: "{collection_name}",
|
||||
prefetch: new List <PrefetchQuery> {
|
||||
new() {
|
||||
Prefetch = {
|
||||
new List <PrefetchQuery> {
|
||||
new() {
|
||||
Query = new float[] { 1, 23, 45, 67 }, // <------------- small byte vector
|
||||
Using = "mrl_byte",
|
||||
Limit = 1000
|
||||
},
|
||||
}
|
||||
},
|
||||
Query = new float[] {0.01f, 0.45f, 0.67f}, // <-- dense vector
|
||||
Using = "full",
|
||||
Limit = 100
|
||||
}
|
||||
},
|
||||
query: new float[][] {
|
||||
[0.1f, 0.2f], // <─┐
|
||||
[0.2f, 0.1f], // < ├─ multi-vector
|
||||
[0.8f, 0.9f] // < ┘
|
||||
},
|
||||
usingVector: "colbert",
|
||||
limit: 10
|
||||
);
|
||||
```
|
||||
|
||||
**Note:** *The multivector feature is not only useful for ColBERT; it can also be used in other ways.*</br>
|
||||
For instance, in e-commerce, you can use multi-vector to store multiple images of the same item. This serves as an alternative to the [group-by](/documentation/concepts/search/#grouping-api) method.
|
||||
|
||||
## Sparse Vectors Compression
|
||||
|
||||
In version 1.9, we introduced the `uint8` [vector datatype](/documentation/concepts/vectors/#datatypes) for sparse vectors, in order to support pre-quantized embeddings from companies like JinaAI and Cohere.
|
||||
This time, we are introducing a new datatype **for both sparse and dense vectors**, as well as a different way of **storing** these vectors.
|
||||
|
||||
**Datatype:** Sparse and dense vectors were previously represented in larger `float32` values, but now they can be turned to the `float16`. `float16` vectors have a lower precision compared to `float32`, which means that there is less numerical accuracy in the vector values - but this is negligible for practical use cases.
|
||||
|
||||
These vectors will use half the memory of regular vectors, which can significantly reduce the footprint of large vector datasets. Operations can be faster due to reduced memory bandwidth requirements and better cache utilization. This can lead to faster vector search operations, especially in memory-bound scenarios.
|
||||
|
||||
When creating a collection, you need to specify the `datatype` upfront:
|
||||
|
||||
```http
|
||||
PUT /collections/{collection_name}
|
||||
{
|
||||
"vectors": {
|
||||
"size": 1024,
|
||||
"distance": "Cosine",
|
||||
"datatype": "float16"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```java
|
||||
import io.qdrant.client.QdrantClient;
|
||||
import io.qdrant.client.QdrantGrpcClient;
|
||||
import io.qdrant.client.grpc.Collections.CreateCollection;
|
||||
import io.qdrant.client.grpc.Collections.Datatype;
|
||||
import io.qdrant.client.grpc.Collections.Distance;
|
||||
import io.qdrant.client.grpc.Collections.VectorParams;
|
||||
import io.qdrant.client.grpc.Collections.VectorsConfig;
|
||||
|
||||
QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build());
|
||||
|
||||
client
|
||||
.createCollectionAsync(
|
||||
CreateCollection.newBuilder()
|
||||
.setCollectionName("{collection_name}")
|
||||
.setVectorsConfig(VectorsConfig.newBuilder()
|
||||
.setParams(VectorParams.newBuilder()
|
||||
.setSize(1024)
|
||||
.setDistance(Distance.Cosine)
|
||||
.setDatatype(Datatype.Float16)
|
||||
.build())
|
||||
.build())
|
||||
.build())
|
||||
.get();
|
||||
```
|
||||
|
||||
```rust
|
||||
use qdrant_client::Qdrant;
|
||||
use qdrant_client::qdrant::{CreateCollectionBuilder, Datatype, Distance, VectorParamsBuilder};
|
||||
|
||||
let client = Qdrant::from_url("http://localhost:6334").build()?;
|
||||
|
||||
client
|
||||
.create_collection(
|
||||
CreateCollectionBuilder::new("{collection_name}").vectors_config(
|
||||
VectorParamsBuilder::new(1024, Distance::Cosine).datatype(Datatype::Float16),
|
||||
),
|
||||
)
|
||||
.await?;
|
||||
```
|
||||
|
||||
```csharp
|
||||
using Qdrant.Client;
|
||||
using Qdrant.Client.Grpc;
|
||||
|
||||
var client = new QdrantClient("localhost", 6334);
|
||||
|
||||
await client.CreateCollectionAsync(
|
||||
collectionName: "{collection_name}",
|
||||
vectorsConfig: new VectorParams {
|
||||
Size = 1024,
|
||||
Distance = Distance.Cosine,
|
||||
Datatype = Datatype.Float16
|
||||
}
|
||||
);
|
||||
```
|
||||
|
||||
**Storage:** On the backend, we implemented bit packing to minimize the bits needed to store data, crucial for handling sparse vectors in applications like machine learning and data compression. For sparse vectors with mostly zeros, this focuses on storing only the indices and values of non-zero elements.
|
||||
|
||||
You will benefit from a more compact storage and higher processing efficiency. This can also lead to reduced dataset sizes for faster processing and lower storage costs in data compression.
|
||||
|
||||
## New Rust Client
|
||||
|
||||
Qdrant’s Rust client has been fully reshaped. It is now more accessible and
|
||||
easier to use. We have focused on putting together a minimalistic API interface.
|
||||
All operations and their types now use the builder pattern, providing an easy
|
||||
and extensible interface, preventing breakage with future updates. See the Rust
|
||||
[ColBERT query](#multivector-support) as great example. Additionally,
|
||||
Rust supports safe concurrent execution, which is crucial for handling multiple
|
||||
simultaneous requests efficiently.
|
||||
|
||||
Documentation got a significant improvement as well. It is much better organized
|
||||
and provides usage examples across the board. Everything links back to our main
|
||||
documentation, making it easier to navigate and find the information you need.
|
||||
|
||||
<p align="center">
|
||||
Visit our
|
||||
<a href="https://docs.rs/qdrant-client/1.10/qdrant_client/">client</a> and
|
||||
<a href="https://docs.rs/qdrant-client/1.10/qdrant_client/struct.Qdrant.html">operations</a> documentation
|
||||
</p>
|
||||
|
||||
## S3 Snapshot Storage
|
||||
Qdrant **Collections**, **Shards** and **Storage** can be backed up with [Snapshots](/documentation/concepts/snapshots/) and saved in case of data loss or other data transfer purposes. These snapshots can be quite large and the resources required to maintain them can result in higher costs. AWS S3 and other S3-compatible implementations like [min.io](https://min.io/) is a great low-cost alternative that can hold snapshots without incurring high costs. It is globally reliable, scalable and resistant to data loss.
|
||||
|
||||
You can configure S3 storage settings in the [config.yaml](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), specifically with `snapshots_storage`.
|
||||
|
||||
For example, to use AWS S3:
|
||||
|
||||
```yaml
|
||||
storage:
|
||||
snapshots_config:
|
||||
# Use 's3' to store snapshots on S3
|
||||
snapshots_storage: s3
|
||||
|
||||
s3_config:
|
||||
# Bucket name
|
||||
bucket: your_bucket_here
|
||||
|
||||
# Bucket region (e.g. eu-central-1)
|
||||
region: your_bucket_region_here
|
||||
|
||||
# Storage access key
|
||||
# Can be specified either here or in the `AWS_ACCESS_KEY_ID` environment variable.
|
||||
access_key: your_access_key_here
|
||||
|
||||
# Storage secret key
|
||||
# Can be specified either here or in the `AWS_SECRET_ACCESS_KEY` environment variable.
|
||||
secret_key: your_secret_key_here
|
||||
```
|
||||
|
||||
*Read more about [S3 snapshot storage](/documentation/concepts/snapshots/#s3) and [configuration](/documentation/guides/configuration/).*
|
||||
|
||||
This integration allows for a more convenient distribution of snapshots. Users of **any S3-compatible object storage** can now benefit from other platform services, such as automated workflows and disaster recovery options. S3's encryption and access control ensure secure storage and regulatory compliance. Additionally, S3 supports performance optimization through various storage classes and efficient data transfer methods, enabling quick and effective snapshot retrieval and management.
|
||||
|
||||
## Issues API
|
||||
Issues API notifies you about potential performance issues and misconfigurations. This powerful new feature allows users (such as database admins) to efficiently manage and track issues directly within the system, ensuring smoother operations and quicker resolutions.
|
||||
|
||||
You can find the Issues button in the top right. When you click the bell icon, a sidebar will open to show ongoing issues.
|
||||
|
||||

|
||||
|
||||
## Minor Improvements
|
||||
|
||||
- Pre-configure collection parameters; quantization, vector storage & replication factor - [#4299](https://github.com/qdrant/qdrant/pull/4299)
|
||||
|
||||
- Overwrite global optimizer configuration for collections. Lets you separate roles for indexing and searching within the single qdrant cluster - [#4317](https://github.com/qdrant/qdrant/pull/4317)
|
||||
|
||||
- Delta encoding and bitpacking compression for sparse vectors reduces memory consumption for sparse vectors by up to 75% - [#4253](https://github.com/qdrant/qdrant/pull/4253), [#4350](https://github.com/qdrant/qdrant/pull/4350)
|
||||
|
||||
@@ -18,7 +18,7 @@ tags:
|
||||
- new features
|
||||
---
|
||||
|
||||
[Qdrant 1.9.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.9.0) This version complements the release of our new managed product [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) with key security features valuable to our enterprise customers, and all those looking to productionize large-scale Generative AI. **Data privacy, system stability and resource optimizations** are always on our mind - so let's see what's new:
|
||||
[Qdrant 1.9.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.9.0) This version complements the release of our new managed product [Qdrant Hybrid Cloud](/hybrid-cloud/) with key security features valuable to our enterprise customers, and all those looking to productionize large-scale Generative AI. **Data privacy, system stability and resource optimizations** are always on our mind - so let's see what's new:
|
||||
|
||||
- **Granular access control:** You can further specify access control levels by using JSON Web Tokens.
|
||||
- **Optimized shard transfers:** The synchronization of shards between nodes is now significantly faster!
|
||||
@@ -34,7 +34,7 @@ Qdrant now supports [granular access control using JSON Web Tokens (JWT)](/docum
|
||||
|
||||

|
||||
|
||||
We highly recommend this feature to enterprises using [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/), as it is tailored to those who need additional control over company data and user access. RBAC empowers administrators to define roles and assign specific privileges to users based on their roles within the organization. In combination with [Hybrid Cloud's data sovereign architecture](/documentation/hybrid-cloud/), this feature reinforces internal security and efficient collaboration by granting access only to relevant resources.
|
||||
We highly recommend this feature to enterprises using [Qdrant Hybrid Cloud](/hybrid-cloud/), as it is tailored to those who need additional control over company data and user access. RBAC empowers administrators to define roles and assign specific privileges to users based on their roles within the organization. In combination with [Hybrid Cloud's data sovereign architecture](/documentation/hybrid-cloud/), this feature reinforces internal security and efficient collaboration by granting access only to relevant resources.
|
||||
|
||||
> **Documentation:** [Read the access level breakdown](/documentation/guides/security/#table-of-access) to see which actions are allowed or denied.
|
||||
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
---
|
||||
title: "Intel’s New CPU Powers Faster Vector Search"
|
||||
draft: false
|
||||
slug: qdrant-cpu-intel-benchmark
|
||||
short_description: "New generation silicon is a game-changer for AI/ML applications."
|
||||
description: "Intel’s 5th gen Xeon processor is made for enterprise-scale operations in vector space. "
|
||||
preview_image: /blog/qdrant-cpu-intel-benchmark/social_preview.jpg
|
||||
social_preview_image: /blog/qdrant-cpu-intel-benchmark/social_preview.jpg
|
||||
date: 2024-05-10T00:00:00-08:00
|
||||
author: David Myriel, Kumar Shivendu
|
||||
featured: false
|
||||
tags:
|
||||
- vector search
|
||||
- intel benchmark
|
||||
- next gen cpu
|
||||
- vector database
|
||||
---
|
||||
|
||||
#### New generation silicon is a game-changer for AI/ML applications
|
||||

|
||||
|
||||
> *Intel’s 5th gen Xeon processor is made for enterprise-scale operations in vector space.*
|
||||
|
||||
Vector search is surging in popularity with institutional customers, and Intel is ready to support the emerging industry. Their latest generation CPU performed exceptionally with Qdrant, a leading vector database used for enterprise AI applications.
|
||||
|
||||
Intel just released the latest Xeon processor (**codename: Emerald Rapids**) for data centers, a market which is expected to grow to $45 billion. Emerald Rapids offers higher-performance computing and significant energy efficiency over previous generations. Compared to the 4th generation Sapphire Rapids, Emerald boosts AI inference performance by up to 42% and makes vector search 38% faster.
|
||||
|
||||
## The CPU of choice for vector database operations
|
||||
|
||||
The latest generation CPU performed exceptionally in tests carried out by Qdrant’s R&D division. Intel’s CPU was stress-tested for query speed, database latency and vector upload time against massive-scale datasets. Results showed that machines with 32 cores were 1.38x faster at running queries than their previous generation counterparts. In this range, Qdrant’s latency also dropped 2.79x when compared to Sapphire.
|
||||
|
||||
Qdrant strongly recommends the use of Intel’s next-gen chips in the 8-64 core range. In addition to being a practical number of cores for most machines in the cloud, this compute capacity will yield the best results with mass-market use cases.
|
||||
|
||||
The CPU affects vector search by influencing the speed and efficiency of mathematical computations. As of recently, companies have started using GPUs to carry large workloads in AI model training and inference. However, for vector search purposes, studies show that CPU architecture is a great fit because it can handle concurrent requests with great ease.
|
||||
|
||||
> *“Vector search is optimized for CPUs. Intel’s new CPU brings even more performance improvement and makes vector operations blazing fast for AI applications. Customers should consider deploying more CPUs instead of GPU compute power to achieve best performance results and reduce costs simultaneously.”*
|
||||
>
|
||||
> - André Zayarni, Qdrant CEO
|
||||
|
||||
## **Why does vector search matter?**
|
||||
|
||||

|
||||
|
||||
Vector search engines empower AI to look deeper into stored data and retrieve strong relevant responses.
|
||||
|
||||
Qdrant’s vector database is key to modern information retrieval and machine learning systems. Those looking to run massive-scale Retrieval Augmented Generation (RAG) solutions need to leverage such semantic search engines in order to generate the best results with their AI products.
|
||||
|
||||
Qdrant is purpose-built to enable developers to store and search for high-dimensional vectors efficiently. It easily integrates with a host of AI/ML tools: Large Language Models (LLM), frameworks such as LangChain, LlamaIndex or Haystack, and service providers like Cohere, OpenAI, and Ollama.
|
||||
|
||||
## Supporting enterprise-scale AI/ML
|
||||
|
||||
The market is preparing for a host of artificial intelligence and machine learning cases, pushing compute to the forefront of the innovation race.
|
||||
|
||||
The main strength of a vector database like Qdrant is that it can consistently support the user way past the prototyping and launch phases. Qdrant’s product is already being used by large enterprises with billions of data points. Such users can go from testing to production almost instantly. Those looking to host large applications might only need up to 18GB RAM to support 1 million OpenAI Vectors. This makes Qdrant the best option for maximizing resource usage and data connection.
|
||||
|
||||
Intel’s latest development is crucial to the future of vector databases. Vector search operations are very CPU-intensive. Therefore, Qdrant relies on the innovations made by chip makers like Intel to offer large-scale support.
|
||||
|
||||
> *“Vector databases are a mainstay in today’s AI/ML toolchain, powering the latest generation of RAG and other Gen AI Applications. In teaming with Qdrant, Intel is helping enterprises deliver cutting-edge Gen-AI solutions and maximize their ROI by leveraging Qdrant’s high-performant and cost-efficient vector similarity search capabilities running on latest Intel Architecture based infrastructure across deployment models.”*
|
||||
>
|
||||
> - Arijit Bandyopadhyay, CTO - Enterprise Analytics & AI, Head of Strategy – Cloud and Enterprise, CSV Group, Intel Corporation
|
||||
|
||||
## Advancing vector search and the role of next-gen CPUs
|
||||
|
||||
Looking ahead, the vector database market is on the cusp of significant growth, particularly for the enterprise market. Developments in CPU technologies, such as those from Intel, are expected to enhance vector search operations by 1) improving processing speeds and 2) boosting retrieval efficiency and quality. This will allow enterprise users to easily manage large and more complex datasets and introduce AI on a global scale.
|
||||
|
||||
As large companies continue to integrate sophisticated AI and machine learning tools, the reliance on robust vector databases is going to increase. This evolution in the market underscores the importance of continuous hardware innovation in meeting the expanding demands of data-intensive applications, with Intel's contributions playing a notable role in shaping the future of enterprise-scale AI/ML solutions.
|
||||
|
||||
## Next steps
|
||||
|
||||
Qdrant is open source and offers a complete SaaS solution, hosted on AWS, GCP, and Azure.
|
||||
|
||||
Getting started is easy, either spin up a [container image](https://hub.docker.com/r/qdrant/qdrant) or start a [free Cloud instance](https://cloud.qdrant.io/login). The documentation covers [adding the data](/documentation/tutorials/bulk-upload/) to your Qdrant instance as well as [creating your indices](/documentation/tutorials/optimize/). We would love to hear about what you are building and please connect with our engineering team on [Github](https://github.com/qdrant/qdrant), [Discord](https://discord.com/invite/tdtYvXjC4h), or [LinkedIn](https://www.linkedin.com/company/qdrant).
|
||||
@@ -0,0 +1,182 @@
|
||||
---
|
||||
title: "Introducing Qdrant Stars: Join Our Ambassador Program!"
|
||||
draft: false
|
||||
slug: qdrant-stars-announcement # Change this slug to your page slug if needed
|
||||
short_description: Qdrant Stars recognizes and supports key contributors to the Qdrant ecosystem through content creation and community leadership. # Change this
|
||||
description: Say hello to the first Qdrant Stars and learn more about our new ambassador program!
|
||||
preview_image: /blog/qdrant-stars-announcement/preview-image.png
|
||||
social_preview_image: /blog/qdrant-stars-announcement/preview-image.png
|
||||
|
||||
date: 2024-05-19T11:57:37-03:00
|
||||
author: Sabrina Aquino
|
||||
featured: true
|
||||
tags:
|
||||
- news
|
||||
- vector search
|
||||
- qdrant
|
||||
- ambassador program
|
||||
- community
|
||||
---
|
||||
|
||||
We're excited to introduce **Qdrant Stars**, our new ambassador program created to recognize and support Qdrant users making a strong impact in the AI and vector search space.
|
||||
|
||||
Whether through innovative content, real-world applications tutorials, educational events, or engaging discussions, they are constantly making vector search more accessible and interesting to explore.
|
||||
|
||||
### 👋 Say hello to the first Qdrant Stars!
|
||||
|
||||
Our inaugural Qdrant Stars are a diverse and talented lineup who have shown exceptional dedication to our community. You might recognize some of their names:
|
||||
|
||||
<div style="display: flex; flex-direction: column;">
|
||||
<div class="qdrant-stars">
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Robert Caulk</h5>
|
||||
<a href="https://www.linkedin.com/in/rcaulk/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Robert LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/robert-caulk-profile.jpeg" alt="Robert Caulk" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Robert is working with a team on <a href="https://asknews.app">AskNews</a> to adaptively enrich, index, and report on over 1 million news articles per day. His team maintains an open-source tool geared toward cluster orchestration <a href="https://flowdapt.ai">Flowdapt</a>, which moves data around highly parallelized production environments. This is why Robert and his team rely on Qdrant for low-latency, scalable, hybrid search across dense and sparse vectors in asynchronous environments.</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I am interested in brainstorming innovative ways to interact with Qdrant vector databases and building presentations that show the power of coupling Flowdapt with Qdrant for large-scale production GenAI applications. I look forward to networking with Qdrant experts and users so that I can learn from their experience.
|
||||
</blockquote>
|
||||
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Joshua Mo</h5>
|
||||
<a href="https://www.linkedin.com/in/joshua-mo-4146aa220/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Josh LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/Josh-Mo-profile.jpg" alt="Josh" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Josh is a Rust developer and DevRel Engineer at <a href="https://shuttle.rs">Shuttle</a>, assisting with user engagement and being a point of contact for first-line information within the community. He's often writing educational content that combines Javascript with Rust and is a coach at Codebar, which is a charity that runs free programming workshops for minority groups within tech.</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I am excited about getting access to Qdrant's new features and contributing to the AI community by demonstrating how those features can be leveraged for production environments.
|
||||
</blockquote>
|
||||
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Nicholas Khami</h5>
|
||||
<a href="https://www.linkedin.com/in/nicholas-khami-5a0a7a135/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Nick LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/ai-headshot-Nick-K.jpg" alt="Nick" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Nick is a founder and product engineer at <a href="https://trieve.ai/">Trieve</a> and has been using Qdrant since late 2022. He has a low level understanding of the Qdrant API, especially the Rust client, and knows a lot about how to make the most of Qdrant on an application level.</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I'm looking forward to be helping folks use lesser known features to enhance and make their projects better!
|
||||
</blockquote>
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Owen Colegrove</h5>
|
||||
<a href="https://www.linkedin.com/in/owencolegrove/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Owen LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/Prof-Owen-Colegrove.jpeg" alt="Owen Colegrove" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Owen Colegrove is the Co-Founder of <a href="https://www.sciphi.ai/">SciPhi</a>, making it easy build, deploy, and scale RAG systems using Qdrant vector search tecnology. He has Ph.D. in Physics and was previously a Quantitative Strategist at Citadel and a Researcher at CERN.</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I'm excited about working together with Qdrant!
|
||||
</blockquote>
|
||||
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Kameshwara Pavan Kumar Mantha</h5>
|
||||
<a href="https://www.linkedin.com/in/kameshwara-pavan-kumar-mantha-91678b21/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Pavan LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/pic-Kameshwara-Pavan-Kumar-Mantha2.jpeg" alt="Kameshwara Pavan" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Kameshwara Pavan is a expert with 14 years of extensive experience in full stack development, cloud solutions, and AI. Specializing in Generative AI and LLMs.
|
||||
Pavan has established himself as a leader in these cutting-edge domains. He holds a Master's in Data Science and a Master's in Computer Applications, and is currently pursuing his PhD.</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
Outside of my professional pursuits, I'm passionate about sharing my knowledge through technical blogging, engaging in technical meetups, and staying active with cycling. I admire the groundbreaking work Qdrant is doing in the industry, and I'm eager to collaborate and learn from the team that drives such exceptional advancements.
|
||||
</blockquote>
|
||||
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Niranjan Akella</h5>
|
||||
<a href="https://www.linkedin.com/in/niranjanakella/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Niranjan LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/nj-Niranjan-Akella.png" alt="Niranjan Akella" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Niranjan is an AI/ML Engineer at <a href="https://www.genesys.com/">Genesys</a> who specializes in building and deploying AI models such as LLMs, Diffusion Models, and Vision Models at scale. He actively shares his projects through content creation and is passionate about applied research, developing custom real-time applications that that serve a greater purpose.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I am a scientist by heart and an AI engineer by profession. I'm always armed to take a leap of faith into the impossible to be come the impossible. I'm excited to explore and venture into Qdrant Stars with some support to build a broader community and develop a sense of completeness among like minded people.
|
||||
</blockquote>
|
||||
|
||||
<div style="display: flex; align-items: center;">
|
||||
<h5>Bojan Jakimovski</h5>
|
||||
<a href="https://www.linkedin.com/in/bojan-jakimovski/" target="_blank"><img src="/blog/qdrant-stars-announcement/In-Blue-40.png" alt="Bojan LinkedIn" style="margin-left:10px; width: 20px; height: 20px;"></a>
|
||||
</div>
|
||||
<div style="display: flex; align-items: center; margin-bottom: 20px;">
|
||||
<img src="/blog/qdrant-stars-announcement/Bojan-preview.jpeg" alt="Bojan Jakimovski" style="width: 200px; height: 200px; object-fit: cover; object-position: center; margin-right: 20px; margin-top: 20px;">
|
||||
<div>
|
||||
<p>Bojan is an Advanced Machine Learning Engineer at <a href="https://www.loka.com/">Loka</a> currently pursuing a Master’s Degree focused on applying AI in Heathcare. He is specializing in Dedicated Computer Systems, with a passion for various technology fields.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
<blockquote>
|
||||
I'm really excited to show the power of the Qdrant as vector database. Especially in some fields where accessing the right data by very fast and efficient way is a must, in fields like Healthcare and Medicine.
|
||||
</blockquote>
|
||||
</div>
|
||||
|
||||
We are happy to welcome this group of people who are deeply committed to advancing vector search technology. We look forward to supporting their vision, and helping them make a bigger impact on the community.
|
||||
|
||||
You can find and chat with them at our [Discord Community](discord.gg/qdrant).
|
||||
|
||||
### Why become a Qdrant Star?
|
||||
|
||||
There are many ways you can benefit from the Qdrant Star Program. Here are just a few:
|
||||
|
||||
##### Exclusive rewards programs
|
||||
|
||||
Celebrate top contributors monthly with special rewards, including exclusive swag and monetary prizes. Quarterly awards for 'Most Innovative Content' and 'Best Tutorial' offer additional prizes.
|
||||
|
||||
##### Early access to new features
|
||||
|
||||
Be the first to explore and write about our latest features and beta products. Participate in product meetings where your ideas and suggestions can directly influence our roadmap.
|
||||
|
||||
##### Conference support
|
||||
|
||||
We love seeing our stars on stage! If you're planning to attend and speak about Qdrant at conferences, we've got you covered. Receive presentation templates, mentorship, and educational materials to help deliver standout conference presentations, with travel expenses covered.
|
||||
|
||||
##### Qdrant Certification
|
||||
|
||||
End the program as a certified Qdrant ambassador and vector search specialist, with provided training resources and a certification test to showcase your expertise.
|
||||
|
||||
### What do Qdrant Stars do?
|
||||
|
||||
As a Qdrant Star, you'll share your knowledge with the community through articles, blogs, tutorials, or demos that highlight the power and versatility of vector search technology - in your own creative way. You'll be a friendly face and a trusted expert in the community, sparking discussions on topics you love and keeping our community active and engaged.
|
||||
|
||||
Love organizing events? You'll have the chance to host meetups, workshops, and other educational gatherings, with all the promotional and logistical support you need to make them a hit. But if large conferences are your thing, we’ll provide the resources and cover your travel expenses so you can focus on delivering an outstanding presentation.
|
||||
|
||||
You'll also have a say in the Qdrant roadmap by giving feedback on new features and participating in product meetings. Qdrant Stars are constantly contributing to the growth and value of the vector search ecosystem.
|
||||
|
||||
### How to join the Qdrant Stars Program
|
||||
|
||||
Are you interested in becoming a Qdrant Star?
|
||||
|
||||
We're on the lookout for individuals who are passionate about vector search technology and looking to make an impact in the AI community.
|
||||
|
||||
If you have a strong understanding of vector search technologies, enjoy creating content, speaking at conferences, and actively engage with our community. If this sounds like you, don't hesitate to apply. We look forward to potentially welcoming you as our next Qdrant Star. [Apply here!](https://forms.gle/q4fkwudDsy16xAZk8)
|
||||
|
||||
Share your journey with vector search technologies and how you plan to contribute further.
|
||||
|
||||
#### Nominate a Qdrant Star
|
||||
|
||||
Do you know someone who could be our next Qdrant Star? Please submit your nomination through our [nomination form](https://forms.gle/n4zv7JRkvnp28qv17), explaining why they're a great fit. Your recommendation could help us find the next standout ambassador.
|
||||
|
||||
#### Learn More
|
||||
|
||||
For detailed information about the program's benefits, activities, and perks, refer to the [Qdrant Stars Handbook](https://qdrant.github.io/qdrant-stars-handbook/).
|
||||
|
||||
To connect with current Stars, ask questions, and stay updated on the latest news and events at Qdrant, [join our Discord community](http://discord.gg/qdrant).
|
||||
+9
-12
@@ -1,14 +1,10 @@
|
||||
---
|
||||
draft: false
|
||||
title: "Qdrant x Dust: How Vector Search helps make work work better - Stan Polu
|
||||
| Vector Space Talks"
|
||||
title: "Unlocking AI Potential: Insights from Stanislas Polu"
|
||||
slug: qdrant-x-dust-vector-search
|
||||
short_description: Stanislas shares insights from his experiences at Stripe and
|
||||
founding his own company, Dust, focusing on AI technology's product layer.
|
||||
description: Stanislas Polu shares insights on integrating SaaS platforms into
|
||||
workflows, reflects on his experiences at Stripe and OpenAI, and discusses his
|
||||
company Dust's focus on enhancing enterprise productivity through tailored AI
|
||||
assistants and their recent switch to Qdrant for database management.
|
||||
description: Explore the dynamic discussion with Stanislas Polu on AI, ML, entrepreneurship, and product development. Gain valuable insights into AI's transformative power.
|
||||
preview_image: /blog/from_cms/stan-polu-cropped.png
|
||||
date: 2024-01-26T16:22:37.487Z
|
||||
author: Demetrios Brinkmann
|
||||
@@ -18,11 +14,13 @@ tags:
|
||||
- Vector Search
|
||||
- OpenAI
|
||||
---
|
||||
|
||||
# Qdrant x Dust: How Vector Search Helps Make Work Better with Stanislas Polu
|
||||
|
||||
> *"We ultimately chose Qdrant due to its open-source nature, strong performance, being written in Rust, comprehensive documentation, and the feeling of control.”*\
|
||||
-- Stanislas Polu
|
||||
>
|
||||
|
||||
|
||||
Stanislas Polu is the Co-Founder and an Engineer at Dust. He had previously sold a company to Stripe and spent 5 years there, seeing them grow from 80 to 3000 people. Then pivoted to research at OpenAI on large language models and mathematical reasoning capabilities. He started Dust 6 months ago to make work work better with LLMs.
|
||||
|
||||
|
||||
@@ -234,19 +232,18 @@ What I want to try.
|
||||
|
||||
|
||||
Demetrios:
|
||||
Okay, the next question that I had is you talked about how benchmarking with the horizontal solution, surprisingly, has been more effective in certain use cases. I'm guessing that's why you got a little bit of love for Qdrant and what we're doing here.
|
||||
Okay, the next question that I had is you talked about how benchmarking with the horizontal solution, surprisingly, has been more effective in certain use cases. I'm guessing that's why you got a little bit of love for [Qdrant](https://qdrant.tech/) and what we're doing here.
|
||||
|
||||
|
||||
Stanislas Polu:
|
||||
Yeah
|
||||
I think the benchmarking was really about quality of models, answers in the.
|
||||
Context of ritual augmented generation.
|
||||
I think the benchmarking was really about quality of models, answers in the context of [retrieval augmented generation](https://qdrant.tech/articles/what-is-rag-in-ai/).
|
||||
So it's not as much as performance, but obviously performance matters, and that's why we love using Qdrants. But I think the main idea of.
|
||||
|
||||
|
||||
Stanislas Polu:
|
||||
What I mentioned is that it's interesting because today the retrieval is noisy, because the embedders are not perfect, which is an interesting point.
|
||||
Sorry, I'm double clicking, but I'll come back. The embedded are really not perfect. Are really not perfect. So that's interesting. When Qdrant release kind of optimization for storage of vectors, they come with obviously warnings that you may have a loss.
|
||||
Sorry, I'm double clicking, but I'll come back. The embedded are really not perfect. Are really not perfect. So that's interesting. When Qdrant release kind of optimization for [storage of vectors](https://qdrant.tech/documentation/concepts/storage/), they come with obviously warnings that you may have a loss.
|
||||
Of precision because of the compression, et cetera, et cetera.
|
||||
And that's funny, like in all kind of retrieval and mental generation world, it really doesn't matter. We take all the performance we can because the loss of precision coming from compression of those vectors at the vector DB level are completely negligible compared to.
|
||||
The holon fuckness of the embedders in.
|
||||
@@ -273,7 +270,7 @@ Saying, oh, I'm working on helping sales find interesting next leads.
|
||||
And you really want to narrow the data exactly where that information lies. And that's where there, we're really relying.
|
||||
Hard on Qdrants as well.
|
||||
So the kind of indexing capabilities on.
|
||||
Top of the vector search, where whenever.
|
||||
Top of the [vector search](https://qdrant.tech/), where whenever.
|
||||
|
||||
|
||||
Stanislas Polu:
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
---
|
||||
title: "QSoC 2024: Announcing Our Interns!"
|
||||
draft: false
|
||||
slug: qsoc24-interns-announcement # Change this slug to your page slug if needed
|
||||
short_description: We are pleased to announce the selection of interns for the inaugural Qdrant Summer of Code (QSoC) program. # Change this
|
||||
description: We are pleased to announce the selection of interns for the inaugural Qdrant Summer of Code (QSoC) program. # Change this
|
||||
preview_image: /blog/qsoc24-interns-announcement/qsoc.jpg # Change this
|
||||
|
||||
social_preview_image: /blog/qsoc24-interns-announcement/qsoc.jpg # Optional image used for link previews
|
||||
title_preview_image: /blog/qsoc24-interns-announcement/qsoc.jpg # Optional image used for blog post title
|
||||
# small_preview_image: /blog/Article-Image.png # Optional image used for small preview in the list of blog posts
|
||||
|
||||
date: 2024-05-08T16:44:22-03:00
|
||||
author: Sabrina Aquino # Change this
|
||||
featured: false # if true, this post will be featured on the blog page
|
||||
tags: # Change this, related by tags posts will be shown on the blog page
|
||||
- QSoC
|
||||
- Qdrant Summer of Code
|
||||
- Google Summer of Code
|
||||
- vector search
|
||||
|
||||
---
|
||||
|
||||
We are excited to announce the interns selected for the inaugural Qdrant Summer of Code (QSoC) program! After receiving many impressive applications, we have chosen two talented individuals to work on the following projects:
|
||||
|
||||
**[Jishan Bhattacharya](https://www.linkedin.com/in/j16n/): WASM-based Dimension Reduction Visualization**
|
||||
|
||||
Jishan will be implementing a dimension reduction algorithm in Rust, compiling it to WebAssembly (WASM), and integrating it with the Qdrant Web UI. This project aims to provide a more efficient and smoother visualization experience, enabling the handling of more data points and higher dimensions efficiently.
|
||||
|
||||
**[Celine Hoang](https://www.linkedin.com/in/celine-h-hoang/): ONNX Cross Encoders in Python**
|
||||
|
||||
Celine Hoang will focus on porting advanced ranking models—specifically Sentence Transformers, ColBERT, and BGE—to the ONNX (Open Neural Network Exchange) format. This project will enhance Qdrant's model support, making it more versatile and efficient in handling complex ranking tasks that are critical for applications such as recommendation engines and search functionalities.
|
||||
|
||||
We look forward to working with Jishan and Celine over the coming months and are excited to see their contributions to the Qdrant project.
|
||||
|
||||
Stay tuned for more updates on the QSoC program and the progress of these projects!
|
||||
|
||||
@@ -77,7 +77,7 @@ The first part of this video explains how caching works. In the second part, you
|
||||
|
||||
## Embrace the Future of AI Data Retrieval
|
||||
|
||||
[Qdrant](https://qdrant.tech/) offers the most flexible way to implement vector search for your RAG and AI applications. You can test out semantic cache on your free Qdrant Cloud instance today! Simply sign up for or log into your [Qdrant Cloud account](https://cloud.qdrant.io/login) and follow our [documentation](/documentation/cloud/).
|
||||
[Qdrant](https://github.com/qdrant/qdrant) offers the most flexible way to implement vector search for your RAG and AI applications. You can test out semantic cache on your free Qdrant Cloud instance today! Simply sign up for or log into your [Qdrant Cloud account](https://cloud.qdrant.io/login) and follow our [documentation](/documentation/cloud/).
|
||||
|
||||
You can also deploy Qdrant locally and manage via our UI. To do this, check our [Hybrid Cloud](/blog/hybrid-cloud/)!
|
||||
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
title: "Qdrant Attains SOC 2 Type II Audit Report"
|
||||
draft: false
|
||||
slug: qdrant-soc2-type2-audit # Change this slug to your page slug if needed
|
||||
short_description: We're proud to announce achieving SOC 2 Type II compliance for Security, Availability, Processing Integrity, Confidentiality, and Privacy.
|
||||
description: We're proud to announce achieving SOC 2 Type II compliance for Security, Availability, and Confidentiality.
|
||||
preview_image: /blog/soc2-type2-report/soc2-preview.jpeg #
|
||||
|
||||
social_preview_image: /blog/soc2-type2-report/soc2-preview.jpeg
|
||||
|
||||
date: 2024-05-23T20:26:20-03:00
|
||||
author: Sabrina Aquino # Change this
|
||||
featured: false # if true, this post will be featured on the blog page
|
||||
tags: # Change this, related by tags posts will be shown on the blog page
|
||||
- soc2
|
||||
- audit
|
||||
- security
|
||||
- confidenciality
|
||||
- data privacy
|
||||
- soc2 type 2
|
||||
|
||||
---
|
||||
|
||||
At Qdrant, we are happy to announce the successful completion our the SOC 2 Type II Audit. This achievement underscores our unwavering commitment to upholding the highest standards of security, availability, and confidentiality for our services and our customers’ data.
|
||||
|
||||
|
||||
## SOC 2 Type II: What Is It?
|
||||
|
||||
SOC 2 Type II certification is an examination of an organization's controls in reference to the American Institute of Certified Public Accountants [(AICPA) Trust Services criteria](https://www.aicpa-cima.com/content/dam/aicpa/interestareas/frc/assuranceadvisoryservices/downloadabledocuments/trust-services-criteria.pdf). It evaluates not only our written policies but also their practical implementation, ensuring alignment between our stated objectives and operational practices. Unlike Type I, which is a snapshot in time, Type II verifies over several months that the company has lived up to those controls. The report represents thorough auditing of our security procedures throughout this examination period: January 1, 2024 to April 7, 2024.
|
||||
|
||||
|
||||
## Key Audit Findings
|
||||
|
||||
The audit ensured with no exceptions noted the effectiveness of our systems and controls on the following Trust Service Criteria:
|
||||
|
||||
|
||||
|
||||
* Security
|
||||
* Confidentiality
|
||||
* Availability
|
||||
|
||||
These certifications are available today and automatically apply to your existing workloads. The full SOC 2 Type II report is available to customers and stakeholders upon request through the [Trust Center](https://app.drata.com/trust/9cbbb75b-0c38-11ee-865f-029d78a187d9).
|
||||
|
||||
|
||||
## Future Compliance
|
||||
|
||||
Going forward, Qdrant will maintain SOC 2 Type II compliance by conducting continuous, annual audits to ensure our security practices remain aligned with industry standards and evolving risks.
|
||||
|
||||
Recognizing the critical importance of data security and the trust our clients place in us, achieving SOC 2 Type II compliance underscores our ongoing commitment to prioritize data protection with the utmost integrity and reliability.
|
||||
|
||||
|
||||
## About Qdrant
|
||||
|
||||
Qdrant is a vector database designed to handle large-scale, high-dimensional data efficiently. It allows for fast and accurate similarity searches in complex datasets. Qdrant strives to achieve seamless and scalable vector search capabilities for various applications.
|
||||
|
||||
For more information about Qdrant and our security practices, please visit our [website](http://qdrant.tech) or [reach out to our team directly](https://qdrant.tech/contact-us/).
|
||||
@@ -1,12 +1,10 @@
|
||||
---
|
||||
draft: false
|
||||
title: Storing multiple vectors per object in Qdrant
|
||||
title: Optimizing Semantic Search by Managing Multiple Vectors
|
||||
slug: storing-multiple-vectors-per-object-in-qdrant
|
||||
short_description: Qdrant's approach to storing multiple vectors per object,
|
||||
unraveling new possibilities in data representation and retrieval.
|
||||
description: Discover how Qdrant continues to push the boundaries of data
|
||||
indexing, providing insights into the practical applications and benefits of
|
||||
this novel vector storage strategy.
|
||||
description: Discover the power of vector storage optimization and learn how to efficiently manage multiple vectors per object for enhanced semantic search capabilities.
|
||||
preview_image: /blog/from_cms/andrey.vasnetsov_a_space_station_with_multiple_attached_modules_853a27c7-05c4-45d2-aebc-700a6d1e79d0.png
|
||||
date: 2022-10-05T10:05:43.329Z
|
||||
author: Kacper Łukawski
|
||||
@@ -18,9 +16,10 @@ tags:
|
||||
- Search
|
||||
- Similarity Search
|
||||
---
|
||||
In a real case scenario, a single object might be described in several different ways. If you run an e-commerce business, then your items will typically have a name, longer textual description and also a bunch of photos. While cooking, you may care about the list of ingredients, and description of the taste but also the recipe and the way your meal is going to look. Up till now, if you wanted to enable semantic search with multiple vectors per object, Qdrant would require you to create separate collections for each vector type, even though they could share some other attributes in a payload. However, since Qdrant 0.10 you are able to store all those vectors together in the same collection and share a single copy of the payload!
|
||||
|
||||
In a real case scenario, a single object might be described in several different ways. If you run an e-commerce business, then your items will typically have a name, longer textual description and also a bunch of photos. While cooking, you may care about the list of ingredients, and description of the taste but also the recipe and the way your meal is going to look. Up till now, if you wanted to enable semantic search with multiple vectors per object, Qdrant would require you to create separate collections for each vector type, even though they could share some other attributes in a payload. However, since Qdrant 0.10 you are able to store all those vectors together in the same collection and share a single copy of the payload!
|
||||
# How to Optimize Vector Storage by Storing Multiple Vectors Per Object
|
||||
|
||||
In a real case scenario, a single object might be described in several different ways. If you run an e-commerce business, then your items will typically have a name, longer textual description and also a bunch of photos. While cooking, you may care about the list of ingredients, and description of the taste but also the recipe and the way your meal is going to look. Up till now, if you wanted to enable [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) with multiple vectors per object, Qdrant would require you to create separate collections for each vector type, even though they could share some other attributes in a payload. However, since Qdrant 0.10 you are able to store all those vectors together in the same collection and share a single copy of the payload!
|
||||
|
||||
Running the new version of Qdrant is as simple as it always was. By running the following command, you are able to set up a single instance that will also expose the HTTP API:
|
||||
|
||||
@@ -177,7 +176,7 @@ The created vectors might be easily put into Qdrant. For the sake of simplicity,
|
||||
|
||||
## Searching with multiple vectors
|
||||
|
||||
If you decided to describe each object with several neural embeddings, then at each search operation you need to provide the vector name along with the embedding, so the engine knows which one to use. The interface of the search operation is pretty straightforward and requires an instance of NamedVector.
|
||||
If you decided to describe each object with several [neural embeddings](https://qdrant.tech/articles/neural-search-tutorial/), then at each search operation you need to provide the vector name along with the [vector embedding](https://qdrant.tech/articles/what-are-embeddings/), so the engine knows which one to use. The interface of the search operation is pretty straightforward and requires an instance of NamedVector.
|
||||
|
||||
```python
|
||||
from qdrant_client.http.models import NamedVector
|
||||
@@ -214,4 +213,11 @@ However, if we use textual description embedding, then the results are slightly
|
||||
|
||||
It is not surprising that a method used for creating neural encoding plays an important role in the search process and its quality. If your data points might be described using several vectors, then the latest release of Qdrant gives you an opportunity to store them together and reuse the payloads, instead of creating several collections and querying them separately.
|
||||
|
||||
### Summary:
|
||||
|
||||
- Qdrant 0.10 introduces efficient vector storage optimization, allowing seamless management of multiple vectors per object within a single collection.
|
||||
- This update streamlines semantic search capabilities by eliminating the need for separate collections for each vector type, enhancing search accuracy and performance.
|
||||
- With Qdrant's new features, users can easily configure vector parameters, including size and distance functions, for each vector type, optimizing search results and user experience.
|
||||
|
||||
|
||||
If you’d like to check out some other examples, please check out our [full notebook](https://gist.github.com/kacperlukawski/961aaa7946f55110abfcd37fbe869b8f) presenting the search results and the whole pipeline implementation.
|
||||
+11
-12
@@ -1,14 +1,11 @@
|
||||
---
|
||||
draft: false
|
||||
title: Superpower your Semantic Search using Vector Database - Nicolas Mauti |
|
||||
title: How to Superpower Your Semantic Search Using a Vector Database
|
||||
Vector Space Talks
|
||||
slug: semantic-search-vector-database
|
||||
short_description: Nicolas Mauti and his team at Malt discusses how they
|
||||
revolutionize the way freelancers connect with projects.
|
||||
description: Nicolas Mauti discusses the improvements to Malt's semantic search
|
||||
capabilities to enhance freelancer and project matching, highlighting the
|
||||
transition to retriever-ranker architecture, implementation of a multilingual
|
||||
encoder model, and the deployment of Qdrant to significantly reduce latency.
|
||||
description: Unlock the secrets of supercharging semantic search with Nicolas Mauti's insights on leveraging vector databases. Discover advanced strategies.
|
||||
preview_image: /blog/from_cms/nicolas-mauti-cropped.png
|
||||
date: 2024-01-09T12:27:18.659Z
|
||||
author: Demetrios Brinkmann
|
||||
@@ -18,6 +15,8 @@ tags:
|
||||
- Retriever-Ranker Architecture
|
||||
- Semantic Search
|
||||
---
|
||||
# How to Superpower Your Semantic Search Using a Vector Database with Nicolas Mauti
|
||||
|
||||
> *"We found a trade off between performance and precision in Qdrant’s that were better for us than what we can found on Elasticsearch.”*\
|
||||
> -- Nicolas Mauti
|
||||
>
|
||||
@@ -34,9 +33,9 @@ Nicolas Mauti, a computer science graduate from INSA Lyon Engineering School, tr
|
||||
|
||||
## **Top Takeaways:**
|
||||
|
||||
Dive into the intricacies of semantic search enhancement with Nicolas Mauti, MLOps Engineer at Malt. Discover how Nicolas and his team at Malt revolutionize the way freelancers connect with projects.
|
||||
Dive into the intricacies of [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) enhancement with Nicolas Mauti, MLOps Engineer at Malt. Discover how Nicolas and his team at Malt revolutionize the way freelancers connect with projects.
|
||||
|
||||
In this episode, Nicolas delves into enhancing semantics search at Malt by implementing a retriever-ranker architecture with multilingual transformer-based models, improving freelancer-project matching through a transition to Qdrant that reduced latency from 10 seconds to 1 second and bolstering the platform's overall performance and scaling capabilities.
|
||||
In this episode, Nicolas delves into enhancing semantics search at Malt by implementing a retriever-ranker architecture with multilingual transformer-based models, improving freelancer-project matching through a transition to [Qdrant](https://qdrant.tech/) that reduced latency from 10 seconds to 1 second and bolstering the platform's overall performance and scaling capabilities.
|
||||
|
||||
5 Keys to Learning from the Episode:
|
||||
|
||||
@@ -134,13 +133,13 @@ Nicolas Mauti:
|
||||
So I think I already talked about this ponds, but yeah, we needed performances. The second ones was about inn quality. As I said before, we cannot do a KnN search, brute force search each time. And so we have to find a way to approximate but to be close enough and to be good enough on these points. And so otherwise we won't be leveraged the performance of our model. And the last one, and I didn't talk a lot about this before, is filtering. Filtering is a big problem for us because we have a lot of filters, of art filters, as I said before. And so if we think about my architecture, we can say, okay, so filtering is not a problem.
|
||||
|
||||
Nicolas Mauti:
|
||||
You can just have a three step process and do filtering, semantic search and then ranking, or do semantic search, filtering and then ranking. But in both cases, you will have some troubles if you do that. The first one is if you want to apply prefiltering. So filtering, semantic search, ranking. If you do that, in fact, you will have, so we'll have this kind of architecture. And if you do that, you will have, in fact, to flag each freelancers before asking the vector database and performing a search, you will have to flag each freelancer whether there could be selected or not. And so with that, you will basically create a binary mask on your freelancers pool. And as the number of freelancers you have will grow, your binary namask will also grow.
|
||||
You can just have a three step process and do filtering, semantic search and then ranking, or do semantic search, filtering and then ranking. But in both cases, you will have some troubles if you do that. The first one is if you want to apply prefiltering. So filtering, semantic search, ranking. If you do that, in fact, you will have, so we'll have this kind of architecture. And if you do that, you will have, in fact, to flag each freelancers before asking the [vector database](https://qdrant.tech/articles/what-is-a-vector-database/) and performing a search, you will have to flag each freelancer whether there could be selected or not. And so with that, you will basically create a binary mask on your freelancers pool. And as the number of freelancers you have will grow, your binary namask will also grow.
|
||||
|
||||
Nicolas Mauti:
|
||||
And so it's not very scalable. And regarding the performance, it will be degraded as your freelancer base grow. And also you will have another problem. A lot of vector database and Qdrants is one of them using hash NSW algorithm to do your inn search. And this kind of algorithm is based on graph. And so if you do that, you will deactivate some nodes in your graph, and so your graph will become disconnected and you won't be able to navigate in your graph. And so your quality of your matching will degrade. So it's definitely not a good idea to apply prefiltering.
|
||||
And so it's not very scalable. And regarding the performance, it will be degraded as your freelancer base grow. And also you will have another problem. A lot of [vector database](https://qdrant.tech/articles/what-is-a-vector-database/) and Qdrants is one of them using hash NSW algorithm to do your inn search. And this kind of algorithm is based on graph. And so if you do that, you will deactivate some nodes in your graph, and so your graph will become disconnected and you won't be able to navigate in your graph. And so your quality of your matching will degrade. So it's definitely not a good idea to apply prefiltering.
|
||||
|
||||
Nicolas Mauti:
|
||||
So, no, if we go to post filtering here, I think the issue is more clear. You will have this kind of architecture. And so, in fact, if you do that, you will have to retrieve a lot of freelancer for your vector database. If you apply some very aggressive filtering and you exclude a lot of freelancer with your filtering, you will have to ask for a lot of freelancer in your vector database and so your performances will be impacted. So filtering is a problem. So we cannot do pre filtering or post filtering. So we had to find a database that do filtering and matching and semantic matching and search at the same time. And so Qdrant is one of them, you have other one in the market.
|
||||
So, no, if we go to post filtering here, I think the issue is more clear. You will have this kind of architecture. And so, in fact, if you do that, you will have to retrieve a lot of freelancer for your [vector database](https://qdrant.tech/articles/what-is-a-vector-database/). If you apply some very aggressive filtering and you exclude a lot of freelancer with your filtering, you will have to ask for a lot of freelancer in your vector database and so your performances will be impacted. So filtering is a problem. So we cannot do pre filtering or post filtering. So we had to find a database that do filtering and matching and semantic matching and search at the same time. And so Qdrant is one of them, you have other one in the market.
|
||||
|
||||
Nicolas Mauti:
|
||||
But in our case, we had one filter that caused us a lot of troubles. And this filter is the geospatial filtering and a few of databases under this filtering, and I think Qdrant is one of them that supports it. But there is not a lot of databases that support them. And we absolutely needed that because we have a local approach and we want to be sure that we recommend freelancer next to the project. And so now that I said all of that, we had three candidates that we tested and we benchmarked them. We had elasticsearch PG vector, that is an extension of PostgreSQL and Qdrants. And on this slide you can see Pycon for example, and Pycon was excluded because of the lack of geospatial filtering. And so we benchmark them regarding the qps.
|
||||
@@ -176,7 +175,7 @@ Demetrios:
|
||||
All right, first off, I want to give a shout out in case there are freelancers that are watching this or looking at this, now is a great time to just join Malt, I think. It seems like it's getting better every day. So I know there's questions that will come through and trickle in, but we've already got one from Luis. What's happening, Luis? He's asking what library or service were you using for Ann before considering Qdrant, in fact.
|
||||
|
||||
Nicolas Mauti:
|
||||
So before that we didn't add any library or service or we were not doing any inn search or semantic search in the way we are doing it right now. We just had one model when we passed the freelancers and the project at the same time in the model, and we got relevancy scoring at the end. And so that's why it was also so slow because you had to constrict each pair and send each pair to your model. And so right now we don't have to do that and so it's much better.
|
||||
So before that we didn't add any library or service or we were not doing any ann search or [semantic searc](https://qdrant.tech/documentation/tutorials/search-beginners/) in the way we are doing it right now. We just had one model when we passed the freelancers and the project at the same time in the model, and we got relevancy scoring at the end. And so that's why it was also so slow because you had to constrict each pair and send each pair to your model. And so right now we don't have to do that and so it's much better.
|
||||
|
||||
Demetrios:
|
||||
Yeah, that makes sense. One question from my side is it took you, I think you said in October you started with the A B test and then in December you rolled it out. What was that last slide that you had?
|
||||
@@ -197,7 +196,7 @@ Nicolas Mauti:
|
||||
Thanks.
|
||||
|
||||
Demetrios:
|
||||
All right, everyone. By the way, in case you want to join us and talk about what you're working on and how you're using Qdrant or what you're doing in the semantic space or semantic search or vector space, all that fun stuff, hit us up. We would love to have you on here. One last question for you, Nicola. Something came through. What indexing method do you use? Is it good for using OpenAI embeddings?
|
||||
All right, everyone. By the way, in case you want to join us and talk about what you're working on and how you're using Qdrant or what you're doing in the semantic space or [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) or vector space, all that fun stuff, hit us up. We would love to have you on here. One last question for you, Nicola. Something came through. What indexing method do you use? Is it good for using OpenAI embeddings?
|
||||
|
||||
Nicolas Mauti:
|
||||
So in our case, we have our own model to build the embeddings.
|
||||
|
||||
@@ -23,7 +23,7 @@ tags:
|
||||
|
||||
Qdrant's vector database quickly grew due to its ability to make Generative AI more effective. On its own, an LLM can be used to build a process-altering invention. With Qdrant, you can turn this invention into a production-level app that brings real business value.
|
||||
|
||||
The use of vector search in GenAI now has a name: **Retrieval Augmented Generation (RAG)**. [In our previous article](https://qdrant.tech/articles/rag-is-dead/), we argued why RAG is an essential component of AI setups, and why large-scale AI can't operate without it. Numerous case studies explain that AI applications are simply too costly and resource-intensive to run using only LLMs.
|
||||
The use of vector search in GenAI now has a name: **Retrieval Augmented Generation (RAG)**. [In our previous article](/articles/rag-is-dead/), we argued why RAG is an essential component of AI setups, and why large-scale AI can't operate without it. Numerous case studies explain that AI applications are simply too costly and resource-intensive to run using only LLMs.
|
||||
|
||||
> Going forward, the solution is to leverage composite systems that use models and vector databases.
|
||||
|
||||
@@ -52,7 +52,7 @@ When supported by LangChain, Qdrant can help you set up effective question-answe
|
||||
Integrating Qdrant and LangChain can revolutionize your AI applications. Let's take a look at what this integration can do for you:
|
||||
|
||||
*Enhance Natural Language Processing (NLP):*
|
||||
LangChain is great for developing question-answering **chatbots**, where Qdrant is used to contextualize and retrieve results for the LLM. We cover this in [our article](https://qdrant.tech/articles/langchain-integration/), and in OpenAI's [cookbook examples](https://cookbook.openai.com/examples/vector_databases/qdrant/qa_with_langchain_qdrant_and_openai) that use LangChain and GPT to process natural language.
|
||||
LangChain is great for developing question-answering **chatbots**, where Qdrant is used to contextualize and retrieve results for the LLM. We cover this in [our article](/articles/langchain-integration/), and in OpenAI's [cookbook examples](https://cookbook.openai.com/examples/vector_databases/qdrant/qa_with_langchain_qdrant_and_openai) that use LangChain and GPT to process natural language.
|
||||
|
||||
*Improve Recommendation Systems:*
|
||||
Food delivery services thrive on indecisive customers. Businesses need to accomodate a multi-aim search process, where customers seek recommendations though semantic search. With LangChain you can build systems for **e-commerce, content sharing, or even dating apps**.
|
||||
@@ -88,19 +88,19 @@ If you are looking to scale up and keep the same level of performance, Qdrant an
|
||||
|
||||
Whether you are building a bank fraud-detection system, RAG for e-commerce, or services for the federal government - you will need to leverage a scalable architecture for your product. Qdrant offers different features to help you considerably increase your application’s performance and lower your hosting costs.
|
||||
|
||||
> Read more about out how we foster [best practices for large-scale deployments](https://qdrant.tech/articles/multitenancy/).
|
||||
> Read more about out how we foster [best practices for large-scale deployments](/articles/multitenancy/).
|
||||
|
||||
## Next Steps
|
||||
|
||||
Now that you know how Qdrant and LangChain can elevate your setup - it's time to try us out.
|
||||
|
||||
- Qdrant is open source and you can [quickstart locally](https://qdrant.tech/documentation/quick-start/), [install it via Docker](https://qdrant.tech/documentation/quick-start/), [or to Kubernetes](https://github.com/qdrant/qdrant-helm/).
|
||||
- Qdrant is open source and you can [quickstart locally](/documentation/quick-start/), [install it via Docker](/documentation/quick-start/), [or to Kubernetes](https://github.com/qdrant/qdrant-helm/).
|
||||
|
||||
- We also offer [a free-tier of Qdrant Cloud](https://cloud.qdrant.io/) for prototyping and testing.
|
||||
|
||||
- For best integration with LangChain, read the [official LangChain documentation](https://python.langchain.com/docs/integrations/vectorstores/qdrant/).
|
||||
|
||||
- For all other cases, [Qdrant documentation](https://qdrant.tech/documentation/integrations/langchain/) is the best place to get there.
|
||||
- For all other cases, [Qdrant documentation](/documentation/integrations/langchain/) is the best place to get there.
|
||||
|
||||
> We offer additional support tailored to your business needs. [Contact us](https://qdrant.to/contact-us) to learn more about implementation strategies and integrations that suit your company.
|
||||
|
||||
|
||||
@@ -0,0 +1,205 @@
|
||||
---
|
||||
title: "What is Vector Similarity? Understanding its Role in AI Applications."
|
||||
draft: false
|
||||
short_description: "An in-depth exploration of vector similarity and its applications in AI."
|
||||
description: "Discover the significance of vector similarity in AI applications and how our vector database revolutionizes similarity search technology for enhanced performance and accuracy."
|
||||
preview_image: /blog/what-is-vector-similarity/social_preview.png
|
||||
social_preview_image: /blog/what-is-vector-similarity/social_preview.png
|
||||
date: 2024-02-24T00:00:00-08:00
|
||||
author: Qdrant Team
|
||||
featured: false
|
||||
tags:
|
||||
- vector search
|
||||
- vector similarity
|
||||
- similarity search
|
||||
- embeddings
|
||||
---
|
||||
|
||||
# Understanding Vector Similarity: Powering Next-Gen AI Applications
|
||||
|
||||
A core function of a wide range of AI applications is to first understand the *meaning* behind a user query, and then provide *relevant* answers to the questions that the user is asking. With increasingly advanced interfaces and applications, this query can be in the form of language, or an image, an audio, video, or other forms of *unstructured* data.
|
||||
|
||||
On an ecommerce platform, a user can, for instance, try to find ‘clothing for a trek’, when they actually want results around ‘waterproof jackets’, or ‘winter socks’. Keyword, or full-text, or even synonym search would fail to provide any response to such a query. Similarly, on a music app, a user might be looking for songs that sound similar to an audio clip they have heard. Or, they might want to look up furniture that has a similar look as the one they saw on a trip.
|
||||
|
||||
## How Does Vector Similarity Work?
|
||||
So, how does an algorithm capture the essence of a user’s query, and then unearth results that are relevant?
|
||||
|
||||
At a high level, here’s how:
|
||||
|
||||
- Unstructured data is first converted into a numerical representation, known as vectors, using a deep-learning model. The goal here is to capture the ‘semantics’ or the key features of this data.
|
||||
- The vectors are then stored in a vector database, along with references to their original data.
|
||||
- When a user performs a query, the query is first converted into its vector representation using the same model. Then search is performed using a metric, to find other vectors which are closest to the query vector.
|
||||
- The list of results returned corresponds to the vectors that were found to be the closest.
|
||||
|
||||
At the heart of all such searches lies the concept of *vector similarity*, which gives us the ability to measure how closely related two data points are, how similar or dissimilar they are, or find other related data points.
|
||||
|
||||
In this document, we will deep-dive into the essence of vector similarity, study how vector similarity search is used in the context of AI, look at some real-world use cases and show you how to leverage the power of vector similarity and vector similarity search for building AI applications.
|
||||
|
||||
## **Understanding Vectors, Vector Spaces and Vector Similarity**
|
||||
|
||||
ML and deep learning models require numerical data as inputs to accomplish their tasks. Therefore, when working with non-numerical data, we first need to convert them into a numerical representation that captures the key features of that data. This is where vectors come in.
|
||||
|
||||
A vector is a set of numbers that represents data, which can be text, image, or audio, or any multidimensional data. Vectors reside in a high-dimensional space, the vector space, where each dimension captures a specific aspect or feature of the data.
|
||||
|
||||
{{< figure width=80% src=/blog/what-is-vector-similarity/working.png caption="Working" >}}
|
||||
|
||||
The number of dimensions of a vector can range from tens or hundreds to thousands, and each dimension is stored as the element of an array. Vectors are, therefore, an array of numbers of fixed length, and in their totality, they encode the key features of the data they represent.
|
||||
|
||||
Vector embeddings are created by AI models, a process known as vectorization. They are then stored in vector stores like Qdrant, which have the capability to rapidly search through vector space, and find similar or dissimilar vectors, cluster them, find related ones, or even the ones which are complete outliers.
|
||||
|
||||
For example, in the case of text data, “coat” and “jacket” have similar meaning, even though the words are completely different. Vector representations of these two words should be such that they lie close to each other in the vector space. The process of measuring their proximity in vector space is vector similarity.
|
||||
|
||||
Vector similarity, therefore, is a measure of how closely related two data points are in a vector space. It quantifies how alike or different two data points are based on their respective vector representations.
|
||||
|
||||
Suppose we have the words "king", "queen" and “apple”. Given a model, words with similar meanings have vectors that are close to each other in the vector space. Vector representations of “king” and “queen” would be, therefore, closer together than "king" and "apple", or “queen” and “apple” due to their semantic relationship. Vector similarity is how you calculate this.
|
||||
|
||||
An extremely powerful aspect of vectors is that they are not limited to representing just text, image or audio. In fact, vector representations can be created out of any kind of data. You can create vector representations of 3D models, for instance. Or for video clips, or molecular structures, or even [protein sequences](https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-019-3220-8).
|
||||
|
||||
There are several methodologies through which vectorization is performed. In creating vector representations of text, for example, the process involves analyzing the text for its linguistic elements using a transformer model. These models essentially learn to capture the essence of the text by dissecting its language components.
|
||||
|
||||
## **How Is Vector Similarity Calculated?**
|
||||
|
||||
There are several ways to calculate the similarity (or distance) between two vectors, which we call metrics. The most popular ones are:
|
||||
|
||||
**Dot Product**: Obtained by multiplying corresponding elements of the vectors and then summing those products. A larger dot product indicates a greater degree of similarity.
|
||||
|
||||
**Cosine Similarity**: Calculated using the dot product of the two vectors divided by the product of their magnitudes (norms). Cosine similarity of 1 implies that the vectors are perfectly aligned, while a value of 0 indicates no similarity. A value of -1 means they are diametrically opposed (or dissimilar).
|
||||
|
||||
**Euclidean Distance**: Assuming two vectors act like arrows in vector space, Euclidean distance calculates the length of the straight line connecting the heads of these two arrows. The smaller the Euclidean distance, the greater the similarity.
|
||||
|
||||
**Manhattan Distance**: Also known as taxicab distance, it is calculated as the total distance between the two vectors in a vector space, if you follow a grid-like path. The smaller the Manhattan distance, the greater the similarity.
|
||||
|
||||
{{< figure width=80% src=/blog/what-is-vector-similarity/products.png caption="Metrics" >}}
|
||||
|
||||
As a rule of thumb, the choice of the best similarity metric depends on how the vectors were encoded.
|
||||
|
||||
Of the four metrics, Cosine Similarity is the most popular.
|
||||
|
||||
## **The Significance of Vector Similarity**
|
||||
|
||||
Vector Similarity is vital in powering machine learning applications. By comparing the vector representation of a query to the vectors of all data points, vector similarity search algorithms can retrieve the most relevant vectors. This helps in building powerful similarity search and recommendation systems, and has numerous applications in image and text analysis, in natural language processing, and in other domains that deal with high-dimensional data.
|
||||
|
||||
Let’s look at some of the key ways in which vector similarity can be leveraged.
|
||||
|
||||
**Image Analysis**
|
||||
|
||||
Once images are converted to their vector representations, vector similarity can help create systems to identify, categorize, and compare them. This can enable powerful reverse image search, facial recognition systems, or can be used for object detection and classification.
|
||||
|
||||
**Text Analysis**
|
||||
|
||||
Vector similarity in text analysis helps in understanding and processing language data. Vectorized text can be used to build semantic search systems, or in document clustering, or plagiarism detection applications.
|
||||
|
||||
**Retrieval Augmented Generation (RAG)**
|
||||
|
||||
Vector similarity can help in representing and comparing linguistic features, from single words to entire documents. This can help build retrieval augmented generation (RAG) applications, where the data is retrieved based on user intent. It also enables nuanced language tasks such as sentiment analysis, synonym detection, language translation, and more.
|
||||
|
||||
**Recommender Systems**
|
||||
|
||||
By converting user preference vectors into item vectors from a dataset, vector similarity can help build semantic search and recommendation systems. This can be utilized in a range of domains such e-commerce or OTT services, where it can help in suggesting relevant products, movies or songs.
|
||||
|
||||
Due to its varied applications, vector similarity has become a critical component in AI tooling. However, implementing it at scale, and in production settings, poses some hard problems. Below we will discuss some of them and explore how Qdrant helps solve these challenges.
|
||||
|
||||
## **Challenges with Vector Similarity Search**
|
||||
|
||||
The biggest challenge in this area comes from what researchers call the "[curse of dimensionality](https://en.wikipedia.org/wiki/Curse_of_dimensionality)." Algorithms like k-d trees may work well for finding exact matches in low dimensions (in 2D or 3D space). However, when you jump to high-dimensional spaces (hundreds or thousands of dimensions, which is common with vector embeddings), these algorithms become impractical. Traditional search methods and OLTP or OLAP databases struggle to handle this curse of dimensionality efficiently.
|
||||
|
||||
This means that building production applications that leverage vector similarity involves navigating several challenges. Here are some of the key challenges to watch out for.
|
||||
|
||||
### Scalability
|
||||
|
||||
Various vector search algorithms were originally developed to handle datasets small enough to be accommodated entirely within the memory of a single computer.
|
||||
|
||||
However, in real-world production settings, the datasets can encompass billions of high-dimensional vectors. As datasets grow, the storage and computational resources required to maintain and search through vector space increases dramatically.
|
||||
|
||||
For building scalable applications, leveraging vector databases that allow for a distributed architecture and have the capabilities of sharding, partitioning and load balancing is crucial.
|
||||
|
||||
### Efficiency
|
||||
|
||||
As the number of dimensions in vectors increases, algorithms that work in lower dimensions become less effective in measuring true similarity. This makes finding nearest neighbors computationally expensive and inaccurate in high-dimensional space.
|
||||
|
||||
For efficient query processing, it is important to choose vector search systems which use indexing techniques that help speed up search through high-dimensional vector space, and reduce latency.
|
||||
|
||||
### Security
|
||||
|
||||
For real-world applications, vector databases frequently house privacy-sensitive data. This can encompass Personally Identifiable Information (PII) in customer records, intellectual property (IP) like proprietary documents, or specialized datasets subject to stringent compliance regulations.
|
||||
|
||||
For data security, the vector search system should offer features that prevent unauthorized access to sensitive information. Also, it should empower organizations to retain data sovereignty, ensuring their data complies with their own regulations and legal requirements, independent of the platform or the cloud provider.
|
||||
|
||||
These are some of the many challenges that developers face when attempting to leverage vector similarity in production applications.
|
||||
|
||||
To address these challenges head-on, we have made several design choices at Qdrant which help power vector search use-cases that go beyond simple CRUD applications.
|
||||
|
||||
## How Qdrant Solves Vector Similarity Search Challenges
|
||||
|
||||
Qdrant is a highly performant and scalable vector search system, developed ground up in Rust. Qdrant leverages Rust’s famed memory efficiency and performance. It supports horizontal scaling, sharding, and replicas, and includes security features like role-based authentication. Additionally, Qdrant can be deployed in various environments, including [hybrid cloud setups](/hybrid-cloud/).
|
||||
|
||||
Here’s how we have taken on some of the key challenges that vector search applications face in production.
|
||||
|
||||
### Efficiency
|
||||
|
||||
Our [choice of Rust](/articles/why-rust/) significantly contributes to the efficiency of Qdrant’s vector similarity search capabilities. Rust’s emphasis on safety and performance, without the need for a garbage collector, helps with better handling of memory and resources. Rust is renowned for its performance and safety features, particularly in concurrent processing, and we leverage it heavily to handle high loads efficiently.
|
||||
|
||||
Also, a key feature of Qdrant is that we leverage both vector and traditional indexes (payload index). This means that vector index helps speed up vector search, while traditional indexes help filter the results.
|
||||
|
||||
The vector index in Qdrant employs the Hierarchical Navigable Small World (HNSW) algorithm for Approximate Nearest Neighbor (ANN) searches, which is one of the fastest algorithms according to [benchmarks](https://github.com/erikbern/ann-benchmarks).
|
||||
|
||||
### Scalability
|
||||
|
||||
For massive datasets and demanding workloads, Qdrant supports [distributed deployment](/documentation/guides/distributed_deployment/) from v0.8.0. In this mode, you can set up a Qdrant cluster and distribute data across multiple nodes, enabling you to maintain high performance and availability even under increased workloads. Clusters support sharding and replication, and harness the Raft consensus algorithm to manage node coordination.
|
||||
|
||||
Qdrant also supports vector [quantization](/documentation/guides/quantization/) to reduce memory footprint and speed up vector similarity searches, making it very effective for large-scale applications where efficient resource management is critical.
|
||||
|
||||
There are three quantization strategies you can choose from - scalar quantization, binary quantization and product quantization - which will help you control the trade-off between storage efficiency, search accuracy and speed.
|
||||
|
||||
### Security
|
||||
|
||||
Qdrant offers several [security features](/documentation/guides/security/) to help protect data and access to the vector store:
|
||||
|
||||
- API Key Authentication: This helps secure API access to Qdrant Cloud with static or read-only API keys.
|
||||
- JWT-Based Access Control: You can also enable more granular access control through JSON Web Tokens (JWT), and opt for restricted access to specific parts of the stored data while building Role-Based Access Control (RBAC).
|
||||
- TLS Encryption: Additionally, you can enable TLS Encryption on data transmission to ensure security of data in transit.
|
||||
|
||||
To help with data sovereignty, Qdrant can be run in a [Hybrid Cloud](/hybrid-cloud/) setup. Hybrid Cloud allows for seamless deployment and management of the vector database across various environments, and integrates Kubernetes clusters into a unified managed service. You can manage these clusters via Qdrant Cloud’s UI while maintaining control over your infrastructure and resources.
|
||||
|
||||
## Optimizing Similarity Search Performance
|
||||
|
||||
In order to achieve top performance in vector similarity searches, Qdrant employs a number of other tactics in addition to the features discussed above.**FastEmbed**: Qdrant supports [FastEmbed](/articles/fastembed/), a lightweight Python library for generating fast and efficient text embeddings. FastEmbed uses quantized transformer models integrated with ONNX Runtime, and is significantly faster than traditional methods of embedding generation.
|
||||
|
||||
**Support for Dense and Sparse Vectors**: Qdrant supports both dense and sparse vector representations. While dense vectors are most common, you may encounter situations where the dataset contains a range of specialized domain-specific keywords. [Sparse vectors](/articles/sparse-vectors/) shine in such scenarios. Sparse vectors are vector representations of data where most elements are zero.
|
||||
|
||||
**Multitenancy**: Qdrant supports [multitenancy](/documentation/guides/multiple-partitions/) by allowing vectors to be partitioned by payload within a single collection. Using this you can isolate each user's data, and avoid creating separate collections for each user. In order to ensure indexing performance, Qdrant also offers ways to bypass the construction of a global vector index, so that you can index vectors for each user independently.
|
||||
|
||||
**IO Optimizations**: If your data doesn’t fit into the memory, it may require storing on disk. To [optimize disk IO performance](/articles/io_uring/), Qdrant offers io_uring based *async uring* storage backend on Linux-based systems. Benchmarks show that it drastically helps reduce operating system overhead from disk IO.
|
||||
|
||||
**Data Integrity**: To ensure data integrity, Qdrant handles data changes in two stages. First, changes are recorded in the Write-Ahead Log (WAL). Then, changes are applied to segments, which store both the latest and individual point versions. In case of abnormal shutdowns, data is restored from WAL.
|
||||
|
||||
**Integrations**: Qdrant has integrations with most popular frameworks, such as LangChain, LlamaIndex, Haystack, Apache Spark, FiftyOne, and more. Qdrant also has several [trusted partners](/blog/hybrid-cloud-launch-partners/) for Hybrid Cloud deployments, such as Oracle Cloud Infrastructure, Red Hat OpenShift, Vultr, OVHcloud, Scaleway, and DigitalOcean.
|
||||
|
||||
We regularly run [benchmarks](/benchmarks/) comparing Qdrant against other vector databases like Elasticsearch, Milvus, and Weaviate. Our benchmarks show that Qdrant consistently achieves the highest requests-per-second (RPS) and lowest latencies across various scenarios, regardless of the precision threshold and metric used.
|
||||
|
||||
## Real-World Use Cases
|
||||
|
||||
Vector similarity is increasingly being used in a wide range of [real-world applications](/use-cases/). In e-commerce, it powers recommendation systems by comparing user behavior vectors to product vectors. In social media, it can enhance content recommendations and user connections by analyzing user interaction vectors. In image-oriented applications, vector similarity search enables reverse image search, similar image clustering, and efficient content-based image retrieval. In healthcare, vector similarity helps in genetic research by comparing DNA sequence vectors to identify similarities and variations. The possibilities are endless.
|
||||
|
||||
A unique example of real-world application of vector similarity is how VISUA uses Qdrant. A leading computer vision platform, VISUA faced two key challenges. First, a rapid and accurate method to identify images and objects within them for reinforcement learning. Second, dealing with the scalability issues of their quality control processes due to the rapid growth in data volume. Their previous quality control, which relied on meta-information and manual reviews, was no longer scalable, which prompted the VISUA team to explore vector databases as a solution.
|
||||
|
||||
After exploring a number of vector databases, VISUA picked Qdrant as the solution of choice. Vector similarity search helped identify similarities and deduplicate large volumes of images, videos, and frames. This allowed VISUA to uniquely represent data and prioritize frames with anomalies for closer examination, which helped scale their quality assurance and reinforcement learning processes. Read our [case study](/blog/case-study-visua/) to learn more.
|
||||
|
||||
## Future Directions and Innovations
|
||||
|
||||
As real-world deployments of vector similarity search technology grows, there are a number of promising directions where this technology is headed.
|
||||
|
||||
We are developing more efficient indexing and search algorithms to handle increasing data volumes and high-dimensional data more effectively. Simultaneously, in case of dynamic datasets, we are pushing to enhance our handling of real-time updates and low-latency search capabilities.
|
||||
|
||||
Qdrant is one of the most secure vector stores out there. However, we are working on bringing more privacy-preserving techniques in vector search implementations to protect sensitive data.
|
||||
|
||||
We have just about witnessed the tip of the iceberg in terms of what vector similarity can achieve. If you are working on an interesting use-case that uses vector similarity, we would like to hear from you.
|
||||
|
||||
## Getting Started with Qdrant
|
||||
|
||||
Ready to implement vector similarity in your AI applications? Explore Qdrant's vector database to enhance your data retrieval and AI capabilities. For additional resources and documentation, visit:
|
||||
|
||||
- [Quick Start Guide](/documentation/quick-start/)
|
||||
- [Documentation](/documentation/)
|
||||
|
||||
We are always available on our [Discord channel](https://qdrant.to/discord) to answer any questions you might have. You can also sign up for our [newsletter](/subscribe/) to stay ahead of the curve.
|
||||
@@ -0,0 +1,11 @@
|
||||
---
|
||||
title: brand-resources
|
||||
description: brand-resources
|
||||
build:
|
||||
render: always
|
||||
cascade:
|
||||
- build:
|
||||
list: local
|
||||
publishResources: false
|
||||
render: never
|
||||
---
|
||||
@@ -0,0 +1,99 @@
|
||||
---
|
||||
logo:
|
||||
title: Our Logo
|
||||
description: "The Qdrant logo represents a paramount expression of our core brand identity. With consistent placement, sizing, clear space, and color usage, our logo affirms its recognition across all platforms."
|
||||
logoCards:
|
||||
- id: 0
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logo.svg
|
||||
alt: Logo Full Color
|
||||
title: Logo Full Color
|
||||
link:
|
||||
url: /img/brand-resources-logos/logo.svg
|
||||
text: Download
|
||||
- id: 1
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logo-black.svg
|
||||
alt: Logo Black
|
||||
title: Logo Black
|
||||
link:
|
||||
url: /img/brand-resources-logos/logo-black.svg
|
||||
text: Download
|
||||
- id: 2
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logo-white.svg
|
||||
alt: Logo White
|
||||
title: Logo White
|
||||
link:
|
||||
url: /img/brand-resources-logos/logo-white.svg
|
||||
text: Download
|
||||
logomarkTitle: Logomark
|
||||
logomarkCards:
|
||||
- id: 0
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logomark.svg
|
||||
alt: Logomark Full Color
|
||||
title: Logomark Full Color
|
||||
link:
|
||||
url: /img/brand-resources-logos/logomark.svg
|
||||
text: Download
|
||||
- id: 1
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logomark-black.svg
|
||||
alt: Logomark Black
|
||||
title: Logomark Black
|
||||
link:
|
||||
url: /img/brand-resources-logos/logomark-black.svg
|
||||
text: Download
|
||||
- id: 2
|
||||
logo:
|
||||
src: /img/brand-resources-logos/logomark-white.svg
|
||||
alt: Logomark White
|
||||
title: Logomark White
|
||||
link:
|
||||
url: /img/brand-resources-logos/logomark-white.svg
|
||||
text: Download
|
||||
colors:
|
||||
title: Colors
|
||||
description: Our brand colors play a crucial role in maintaining a cohesive visual identity. The careful balance of these colors ensures a consistent and impactful representation of Qdrant, reinforcing our commitment to excellence and precision in every aspect of our work.
|
||||
cards:
|
||||
- id: 0
|
||||
name: Amaranth
|
||||
type: HEX
|
||||
code: "DC244C"
|
||||
- id: 1
|
||||
name: Blue
|
||||
type: HEX
|
||||
code: "2F6FF0"
|
||||
- id: 2
|
||||
name: Violet
|
||||
type: HEX
|
||||
code: "8547FF"
|
||||
- id: 3
|
||||
name: Teal
|
||||
type: HEX
|
||||
code: "038585"
|
||||
- id: 4
|
||||
name: Black
|
||||
type: HEX
|
||||
code: "090E1A"
|
||||
- id: 5
|
||||
name: White
|
||||
type: HEX
|
||||
code: "FFFFFF"
|
||||
typography:
|
||||
title: Typography
|
||||
description: Main typography is Satoshi, this is employed for both UI and marketing purposes. Headlines are set in Bold (600), while text is rendered in Medium (500).
|
||||
example: AaBb
|
||||
specimen: "ABCDEFGHIJKLMNOPQRSTUVWXYZ<br>abcdefghijklmnopqrstuvwxyz<br>0123456789 !@#$%^&*()"
|
||||
link:
|
||||
url: https://api.fontshare.com/v2/fonts/download/satoshi
|
||||
text: Download
|
||||
trademarks:
|
||||
title: Trademarks
|
||||
description: All features associated with the Qdrant brand are safeguarded by relevant trademark, copyright, and intellectual property regulations. Utilization of the Qdrant trademark must adhere to the specified Qdrant Trademark Standards for Use.<br><br>Should you require clarification or seek permission to utilize these resources, feel free to reach out to us at
|
||||
link:
|
||||
url: "mailto:info@qdrant.com"
|
||||
text: info@qdrant.com.
|
||||
sitemapExclude: true
|
||||
---
|
||||
@@ -0,0 +1,18 @@
|
||||
---
|
||||
title: Qdrant Brand Resources
|
||||
buttons:
|
||||
- id: 0
|
||||
url: "#logo"
|
||||
text: Logo
|
||||
- id: 1
|
||||
url: "#colors"
|
||||
text: Colors
|
||||
- id: 2
|
||||
url: "#typography"
|
||||
text: Typography
|
||||
- id: 3
|
||||
url: "#trademarks"
|
||||
text: Trademarks
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
---
|
||||
title: Community
|
||||
description: Community
|
||||
build:
|
||||
render: always
|
||||
cascade:
|
||||
- build:
|
||||
list: local
|
||||
publishResources: false
|
||||
render: never
|
||||
---
|
||||
@@ -0,0 +1,78 @@
|
||||
---
|
||||
title: Discover our Programs
|
||||
resources:
|
||||
- id: 0
|
||||
title: Qdrant Stars
|
||||
description: Qdrant Stars are our top contributors, organizers, and evangelists. Learn more about how you can become a Star.
|
||||
link:
|
||||
text: Learn More
|
||||
url: /blog/qdrant-stars-announcement/
|
||||
image:
|
||||
src: /img/community-features/qdrant-stars.svg
|
||||
alt: Avatar
|
||||
- id: 1
|
||||
title: Discord
|
||||
description: Chat in real-time with the Qdrant team and community members.
|
||||
link:
|
||||
text: Join our Discord
|
||||
url: https://discord.gg/qdrant
|
||||
image:
|
||||
src: /img/community-features/discord.svg
|
||||
alt: Avatar
|
||||
- id: 2
|
||||
title: Community Blog
|
||||
description: Learn all the latest tips and tricks in the AI space through our community blog.
|
||||
link:
|
||||
text: Visit our Blog
|
||||
url: /blog/
|
||||
image:
|
||||
src: /img/community-features/community-blog.svg
|
||||
alt: Avatar
|
||||
- id: 3
|
||||
title: Vector Space Talks
|
||||
description: Weekly tech talks with Qdrant users and industry experts.
|
||||
link:
|
||||
text: Learn More
|
||||
url: https://www.youtube.com/watch?v=4aUq5VnR_VI&list=PL9IXkWSmb36_eANzd_sKeQ3tXbFiEGEWn&pp=iAQB
|
||||
image:
|
||||
src: /img/community-features/vector-space-talks.svg
|
||||
alt: Avatar
|
||||
features:
|
||||
- id: 0
|
||||
icon:
|
||||
src: /icons/outline/documentation-blue.svg
|
||||
alt: Documentation
|
||||
title: Documentation
|
||||
description: Docs carefully crafted to support developers and decision-makers learning about Qdrant features.
|
||||
link:
|
||||
text: Read More
|
||||
url: /documentation/
|
||||
- id: 1
|
||||
icon:
|
||||
src: /icons/outline/guide-blue.svg
|
||||
alt: Guide
|
||||
title: Contributors Guide
|
||||
description: Whatever your strengths are, we got you covered. Learn more about how to contribute to Qdrant.
|
||||
link:
|
||||
text: Learn More
|
||||
url: https://github.com/qdrant/qdrant/blob/master/CONTRIBUTING.md
|
||||
- id: 2
|
||||
icon:
|
||||
src: /icons/outline/handshake-blue.svg
|
||||
alt: Partners
|
||||
title: Partners
|
||||
description: Technology partners and applications that support Qdrant.
|
||||
link:
|
||||
text: Learn More
|
||||
url: /partners/
|
||||
- id: 3
|
||||
icon:
|
||||
src: /icons/outline/mail-blue.svg
|
||||
alt: Newsletter
|
||||
title: Newsletter
|
||||
description: Stay up to date with all the latest Qdrant news
|
||||
link:
|
||||
text: Learn More
|
||||
url: /subscribe/
|
||||
sitemapExclude: true
|
||||
---
|
||||
@@ -0,0 +1,14 @@
|
||||
---
|
||||
title: Welcome to the Qdrant Community
|
||||
description: Connect with over 30,000 community members, get access to educational resources, and stay up to date on all news and discussions about Qdrant and the vector database space.
|
||||
image:
|
||||
src: /img/community-hero.svg
|
||||
srcMobile: /img/mobile/community-hero.svg
|
||||
alt: Community
|
||||
button:
|
||||
text: Join our Discord
|
||||
url: https://discord.gg/qdrant
|
||||
about: Get access to educational resources, and stay up to date on all news and discussions about Qdrant and the vector database space.
|
||||
sitemapExclude: true
|
||||
---
|
||||
|
||||
@@ -0,0 +1,47 @@
|
||||
---
|
||||
title: Love from our community
|
||||
testimonials:
|
||||
- id: 0
|
||||
name: Owen Colegrove
|
||||
nickname: "@ocolegro"
|
||||
avatar:
|
||||
src: /img/customers/owen-colegrove.svg
|
||||
alt: Avatar
|
||||
text: qurant has been amazing!
|
||||
- id: 1
|
||||
name: Darren
|
||||
nickname: "@darrenangle"
|
||||
avatar:
|
||||
src: /img/customers/darren.svg
|
||||
alt: Avatar
|
||||
text: qdrant is so fast I'm using Rust for all future projects goodnight everyone
|
||||
- id: 2
|
||||
name: Greg Schoeninger
|
||||
nickname: "@gregschoeninger"
|
||||
avatar:
|
||||
src: /img/customers/greg-schoeninger.svg
|
||||
alt: Avatar
|
||||
text: Indexing millions of embeddings into <span>@qdrant_engine</span> has been the smoothest experience I've had so far with a vector db. Team Rustacian all the way 🦀
|
||||
- id: 3
|
||||
name: Ifioravanti
|
||||
nickname: "@ivanfioravanti"
|
||||
avatar:
|
||||
src: /img/customers/ifioravanti.svg
|
||||
alt: Avatar
|
||||
text: <span>@qdrant_engine</span> is ultra super powerful! Combine it to <span>@LangChainAI</span> and you have a super productivity boost for your AI projects ⏩⏩⏩
|
||||
- id: 4
|
||||
name: sengpt
|
||||
nickname: "@sengpt"
|
||||
avatar:
|
||||
src: /img/customers/sengpt.svg
|
||||
alt: Avatar
|
||||
text: Thank you, Qdrant is awesome
|
||||
- id: 4
|
||||
name: Owen Colegrove
|
||||
nickname: "@ocolegro"
|
||||
avatar:
|
||||
src: /img/customers/owen-colegrove.svg
|
||||
alt: Avatar
|
||||
text: that sounds good to me, big fan of qdrant.
|
||||
sitemapExclude: true
|
||||
---
|
||||
@@ -0,0 +1,30 @@
|
||||
---
|
||||
title: Qdrant Hybrid Cloud
|
||||
salesTitle: Hybrid Cloud
|
||||
description: Bring your own Kubernetes clusters from any cloud provider, on-premise infrastructure, or edge locations and connect them to the Managed Cloud.
|
||||
cards:
|
||||
- id: 0
|
||||
icon: /icons/outline/separate-blue.svg
|
||||
title: Deployment Flexibility
|
||||
description: Use your existing infrastructure, whether it be on cloud platforms, on-premise setups, or even at edge locations.
|
||||
- id: 1
|
||||
icon: /icons/outline/money-growth-blue.svg
|
||||
title: Unmatched Cost Advantage
|
||||
description: Maximum deployment flexibility to leverage the best available resources, in the cloud or on-premise.
|
||||
- id: 2
|
||||
icon: /icons/outline/switches-blue.svg
|
||||
title: Transparent Control
|
||||
description: Fully managed experience for your Qdrant clusters, while your data remains exclusively yours.
|
||||
form:
|
||||
title: Connect with us
|
||||
# description:
|
||||
id: contact-sales-form
|
||||
hubspotFormOptions: '{
|
||||
"region": "eu1",
|
||||
"portalId": "139603372",
|
||||
"formId": "f583c7ea-15ff-4c57-9859-650b8f34f5d3",
|
||||
"submitButtonClass": "button button_contained",
|
||||
}'
|
||||
logosSectionTitle: Qdrant is trusted by top-tier enterprises
|
||||
---
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
salesTitle: Qdrant Enterprise Solutions
|
||||
description: Our Managed Cloud, Hybrid Cloud, and Private Cloud solutions offer flexible deployment options for top-tier data privacy.
|
||||
cards:
|
||||
- id: 0
|
||||
icon: /icons/outline/cloud-managed-blue.svg
|
||||
title: Managed Cloud
|
||||
description: Qdrant Cloud provides optimal flexibility and offers a suite of features focused on efficient and scalable vector search - fully managed. Available on AWS, Google Cloud, and Azure.
|
||||
- id: 1
|
||||
icon: /icons/outline/cloud-hybrid-violet.svg
|
||||
title: Hybrid Cloud
|
||||
description: Bring your own Kubernetes clusters from any cloud provider, on-premise infrastructure, or edge locations and connect them to the Managed Cloud.
|
||||
- id: 2
|
||||
icon: /icons/outline/cloud-private-teal.svg
|
||||
title: Private Cloud
|
||||
description: Deploy Qdrant in your own infrastructure.
|
||||
form:
|
||||
title: Connect with us
|
||||
# description:
|
||||
id: contact-sales-form
|
||||
hubspotFormOptions: '{
|
||||
"region": "eu1",
|
||||
"portalId": "139603372",
|
||||
"formId": "fc7a9f1d-9d41-418d-a9cc-ef9c5fb9b207",
|
||||
"submitButtonClass": "button button_contained",
|
||||
}'
|
||||
logosSectionTitle: Qdrant is trusted by top-tier enterprises
|
||||
---
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
title: Contact Qdrant
|
||||
description: Please let us know how we can help and we will get in touch with you soon.
|
||||
cards:
|
||||
- id: 0
|
||||
icon: /icons/outline/comments-violet.svg
|
||||
title: Qdrant Cloud Support
|
||||
description: For questions or issues with Qdrant Cloud, please report to
|
||||
mailLink:
|
||||
text: support@qdrant.io
|
||||
href: support@qdrant.io
|
||||
- id: 1
|
||||
icon: /icons/outline/discord-blue.svg
|
||||
title: Developer Support
|
||||
description: For developer questions about Qdrant usage, please join our
|
||||
link:
|
||||
text: Discord Server
|
||||
href: https://qdrant.to/discord
|
||||
form:
|
||||
id: contact-us-form
|
||||
title: Talk to our Team
|
||||
hubspotFormOptions: '{
|
||||
"region": "eu1",
|
||||
"portalId": "139603372",
|
||||
"formId": "814b303f-2f24-460a-8a81-367146d98786",
|
||||
"submitButtonClass": "button button_contained",
|
||||
}'
|
||||
---
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
---
|
||||
title: Customers
|
||||
description: Customers
|
||||
build:
|
||||
render: always
|
||||
cascade:
|
||||
- build:
|
||||
list: local
|
||||
publishResources: false
|
||||
render: never
|
||||
---
|
||||
@@ -0,0 +1,51 @@
|
||||
---
|
||||
title: Customers
|
||||
description: Learn how Qdrant powers thousands of top AI solutions that require vector search with unparalleled efficiency, performance and massive-scale data processing.
|
||||
caseStudy:
|
||||
logo:
|
||||
src: /img/customers-case-studies/customer-logo.svg
|
||||
alt: Logo
|
||||
title: Recommendation Engine with Qdrant Vector Database
|
||||
description: Dailymotion leverages Qdrant to optimize its <b>video recommendation engine</b>, managing over 420 million videos and processing 13 million recommendations daily. With this, Dailymotion was able to <b>reduced content processing times from hours to minutes</b> and <b>increased user interactions and click-through rates by more than 3x.</b>
|
||||
link:
|
||||
text: Read Case Study
|
||||
url: /blog/case-study-dailymotion/
|
||||
image:
|
||||
src: /img/customers-case-studies/case-study.png
|
||||
alt: Preview
|
||||
cases:
|
||||
- id: 0
|
||||
logo:
|
||||
src: /img/customers-case-studies/visua.svg
|
||||
alt: Visua Logo
|
||||
image:
|
||||
src: /img/customers-case-studies/case-visua.png
|
||||
alt: The hands of a person in a medical gown holding a tablet against the background of a pharmacy shop
|
||||
title: VISUA improves quality control process for computer vision with anomaly detection by 10x.
|
||||
link:
|
||||
text: Read Story
|
||||
url: /blog/case-study-visua/
|
||||
- id: 1
|
||||
logo:
|
||||
src: /img/customers-case-studies/dust.svg
|
||||
alt: Dust Logo
|
||||
image:
|
||||
src: /img/customers-case-studies/case-dust.png
|
||||
alt: A man in a jeans shirt is holding a smartphone, only his hands are visible. In the foreground, there is an image of a robot surrounded by chat and sound waves.
|
||||
title: Dust uses Qdrant for RAG, achieving millisecond retrieval, reducing costs by 50%, and boosting scalability.
|
||||
link:
|
||||
text: Read Story
|
||||
url: /blog/dust-and-qdrant/
|
||||
- id: 2
|
||||
logo:
|
||||
src: /img/customers-case-studies/iris-agent.svg
|
||||
alt: Logo
|
||||
image:
|
||||
src: /img/customers-case-studies/case-iris-agent.png
|
||||
alt: Hands holding a smartphone, styled smartphone interface visualisation in the foreground. First-person view
|
||||
title: IrisAgent uses Qdrant for RAG to automate support, and improve resolution times, transforming customer service.
|
||||
link:
|
||||
text: Read Story
|
||||
url: /blog/iris-agent-qdrant/
|
||||
sitemapExclude: true
|
||||
---
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user