diff --git a/.github/workflows/check-dead-links.yml b/.github/workflows/check-dead-links.yml
index 4d374e0e0..27e18696c 100644
--- a/.github/workflows/check-dead-links.yml
+++ b/.github/workflows/check-dead-links.yml
@@ -10,7 +10,7 @@ jobs:
linkChecker:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Setup Hugo
uses: peaceiris/actions-hugo@2752ce1d29631191ea3f27c23495fa06139a5b78 # v3.2.1
with:
@@ -23,7 +23,7 @@ jobs:
run: bash -x ./install-and-build.sh
- name: Link Checker
id: lychee
- uses: lycheeverse/lychee-action@8646ba30535128ac92d33dfc9133794bfdd9b411 # v2.8.0
+ uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
with:
args: 'qdrant-landing/public'
fail: false
diff --git a/.github/workflows/generate-preview-images.yml b/.github/workflows/generate-preview-images.yml
index 982560ac2..6b5350056 100644
--- a/.github/workflows/generate-preview-images.yml
+++ b/.github/workflows/generate-preview-images.yml
@@ -12,7 +12,7 @@ jobs:
sync:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
diff --git a/.github/workflows/internal-dead-links.yml b/.github/workflows/internal-dead-links.yml
index de17c3dbb..1f4163971 100644
--- a/.github/workflows/internal-dead-links.yml
+++ b/.github/workflows/internal-dead-links.yml
@@ -11,7 +11,7 @@ jobs:
linkChecker:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Setup Hugo
uses: peaceiris/actions-hugo@2752ce1d29631191ea3f27c23495fa06139a5b78 # v3.2.1
with:
@@ -34,7 +34,7 @@ jobs:
done
- name: Internal Links Check
id: lychee
- uses: lycheeverse/lychee-action@8646ba30535128ac92d33dfc9133794bfdd9b411 # v2.8.0
+ uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0
with:
args: --max-redirects 0 --exclude '.*' --include '^http://localhost:1314/[^%]+$' --base http://localhost:1314/ qdrant-landing/public/
fail: true
diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml
index 6125ebbb4..8ffd16f16 100644
--- a/.github/workflows/main.yml
+++ b/.github/workflows/main.yml
@@ -7,7 +7,7 @@ jobs:
sync:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
diff --git a/.github/workflows/snippets.yml b/.github/workflows/snippets.yml
index e13492b4f..24ac6042a 100644
--- a/.github/workflows/snippets.yml
+++ b/.github/workflows/snippets.yml
@@ -10,7 +10,7 @@ jobs:
convert-snippets:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Convert runnable snippets to markdown files
run: automation/snippets/generate-md.py
@@ -25,9 +25,12 @@ jobs:
fi
typecheck-snippets:
+ if: >
+ !(contains(github.event.pull_request.labels.*.name, 'skip snippet check') && github.event.pull_request.base.ref != 'master')
runs-on: ubuntu-latest
+ needs: [convert-snippets]
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Build snippet-checker Docker image
run: docker build -t snippet-checker automation/snippets/docker
diff --git a/.github/workflows/update-stats.yml b/.github/workflows/update-stats.yml
index fba7b651e..d01e66283 100644
--- a/.github/workflows/update-stats.yml
+++ b/.github/workflows/update-stats.yml
@@ -10,14 +10,14 @@ jobs:
update:
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
- name: Run github-stars update script
run: |
bash -x automation/update-stats.sh
- - uses: stefanzweifel/git-auto-commit-action@04702edda442b2e678b25b537cec683a1493fcb9 # v7.1.0
+ - uses: stefanzweifel/git-auto-commit-action@4a55954c782fc1ea30b9056cd3e7a2b40ca8887d # v7.2.0
with:
commit_message: "Update GitHub Stars"
commit_user_name: "GitHub Actions"
diff --git a/.gitignore b/.gitignore
index 4dffe010f..82d600d5e 100644
--- a/.gitignore
+++ b/.gitignore
@@ -8,4 +8,5 @@ package-lock.json
dart-sass/
.env
-.venv/
\ No newline at end of file
+.venv/
+.agents/
diff --git a/README.md b/README.md
index 0f2ccd9d7..885ab5b1a 100644
--- a/README.md
+++ b/README.md
@@ -25,6 +25,9 @@
- [Images](#images)
- [Important notes](#important-notes)
- [Agenda](#agenda)
+ - [Demo](#demo)
+ - [Add a demo](#add-a-demo)
+ - [Add a filter](#add-a-filter)
- [Shortcodes 🧩🧩🧩](#shortcodes-)
- [Built-in shortcodes](#built-in-shortcodes)
- [Custom shortcodes](#custom-shortcodes)
@@ -363,6 +366,56 @@ Optional talk parameters:
The layout lives at `themes/qdrant-2024/layouts/agenda/single.html` and styles at `themes/qdrant-2024/assets/css/partials/_agenda.scss`.
+## Demo
+
+Demos and filters for the `/demo` page live in `qdrant-landing/content/demo/items/_index.md`. Edit that file only — no template changes needed for new demos or filters.
+
+### Add a demo
+
+Append an entry under `demos:`:
+
+```yaml
+demos:
+ - id: my-new-demo # unique slug
+ title: My New Demo
+ description: Short description shown on the card.
+ category: Semantic Search # must match a filter field (see below)
+ image: /img/demos/demo-0.png # optional; omit for a placeholder
+ github: https://github.com/org/repo # optional; icon link on the card
+ weight: 10 # optional; same rules as Hugo page weight
+ link:
+ text: View Demo
+ url: https://example.com/
+```
+
+`weight` follows Hugo’s built-in page weight rules: use a non-zero integer; lighter items float to the top, heavier sink to the bottom; missing or `0` weight is placed at the end. Ties break by title.
+
+Put card images in `themes/qdrant-2024/static/img/demos/`. Provide a PNG and a matching WebP at **800×296px** (same basename, e.g. `demo-0.png` + `demo-0.webp`). Only list the PNG file in the markdown; the picture partial swaps the extension to serve WebP when available.
+
+### Add a filter
+
+Each filter needs a `key` that matches a field on every demo, and a `label` for the sidebar. Filter options are collected automatically from demo values unless you set `values` explicitly.
+
+```yaml
+filters:
+ - key: category
+ label: Categories
+ - key: industry # new filter
+ label: Industries
+
+demos:
+ - id: my-new-demo
+ title: My New Demo
+ description: Short description shown on the card.
+ category: Semantic Search
+ industry: Healthcare # same key as the new filter
+ link:
+ text: View Demo
+ url: https://example.com/
+```
+
+Optional: `batchSize` controls how many cards show before “View More” (default `8`).
+
## Shortcodes 🧩🧩🧩
Hugo lets you use built-in and custom shortcodes to simplify the creation of content. Meanwhile, **keep in mind that shortcodes make the content less portable**. If you decide to move the content to another platform, you'll need to rewrite the shortcodes. **Avoid to overuse them.**
diff --git a/automation/snippets/README.md b/automation/snippets/README.md
index bbff791b9..2568b126b 100644
--- a/automation/snippets/README.md
+++ b/automation/snippets/README.md
@@ -201,8 +201,16 @@ Each supported language has:
## Quirks
-Sometimes `mypy` (python typechecker) complains at valid code.
-Place this comment at the top of the file to silence it:
-```python
-# mypy: disable-error-code="arg-type"
-```
+- Sometimes `mypy` (python typechecker) complains at valid code.
+ Place this comment at the top of the file to silence it:
+ ```python
+ # mypy: disable-error-code="arg-type"
+ ```
+- On MacOS, if you see Java compile errors like this:
+ ```
+ > java.io.IOException: Cannot run program "/workspace/automation/snippets/cache/.gradle/caches/modules-2/files-2.1/com.google.protobuf/protoc/3.25.5/601137f5367caaf202a28e3844dd6dbbc77b19af/protoc-3.25.5-linux-x86_64.exe": Exec failed, error: 13 (Permission denied)
+ ```
+ Explicitly set the executable bit on the reported file, for example (from `automation/snippets`):
+ ```bash
+ chmod a+x ./cache/.gradle/caches/modules-2/files-2.1/com.google.protobuf/protoc/3.25.5/601137f5367caaf202a28e3844dd6dbbc77b19af/protoc-3.25.5-linux-x86_64.exe
+ ```
diff --git a/automation/snippets/check.py b/automation/snippets/check.py
index 8bc0e7b23..9e8898c95 100755
--- a/automation/snippets/check.py
+++ b/automation/snippets/check.py
@@ -10,7 +10,7 @@ import types
import typing
from pathlib import Path
-from lib import SNIPPETS_DIR, CollectedSnippetsType, collect_snippets, log
+from lib import SNIPPETS_DIR, CollectedSnippetsType, collect_snippets, extract_code, log
from lib.languages import ALL_LANGUAGES, CompileResult, Language, parse_languages
@@ -85,6 +85,29 @@ def build_and_run(
spec.loader.exec_module(mod)
test_modules[snippet_dir] = mod
+ log("Syntax check stage")
+ for lang in snippets_by_lang:
+ if not lang.SUPPORTS_SYNTAX_CHECK:
+ log(
+ f"· Cannot check syntax of generated {lang.NAME} markdown "
+ "(no syntax-only parser available) - please review it manually"
+ )
+ continue
+
+ log(f"· Checking syntax of generated {lang.NAME} markdown")
+ for snippet_dir in snippets:
+ generated_dir = snippet_dir / "generated"
+ if not generated_dir.is_dir():
+ continue
+ for md_fname in sorted(generated_dir.rglob(f"{lang.NAME}.md")):
+ try:
+ lang.check_syntax(extract_code(lang, md_fname))
+ except Exception as e:
+ log(f"· · Syntax check failed for {md_fname}")
+ errors.append(f"Syntax check failed for {md_fname}")
+ if not isinstance(e, subprocess.CalledProcessError):
+ traceback.print_exc()
+
log("Compile stage")
for lang, fnames in snippets_by_lang.items():
log(f"· Compiling {lang.NAME} snippets")
diff --git a/automation/snippets/lib/__init__.py b/automation/snippets/lib/__init__.py
index 0122c7ec2..02dec2577 100644
--- a/automation/snippets/lib/__init__.py
+++ b/automation/snippets/lib/__init__.py
@@ -1,4 +1,5 @@
import os
+import re
import sys
import typing
from pathlib import Path
@@ -39,6 +40,13 @@ Example:
}
"""
+def extract_code(language: type[Language], file: Path) -> str:
+ content = file.read_text()
+ regex = re.compile(rf"```{language.NAME}\r?\n([\s\S]*?)\r?\n```")
+ matches = regex.findall(content)
+ if len(matches) == 0:
+ raise RuntimeError("No code find in markdown file")
+ return matches[0]
def collect_snippets(
bases: list[Path],
diff --git a/automation/snippets/lib/languages/base.py b/automation/snippets/lib/languages/base.py
index a174b3aed..f8cdc73c0 100644
--- a/automation/snippets/lib/languages/base.py
+++ b/automation/snippets/lib/languages/base.py
@@ -17,6 +17,9 @@ class Language:
SNIPPET_FILENAME: str
"""Filename of the snippet in this language, e.g., 'python.py'."""
+ SUPPORTS_SYNTAX_CHECK: bool = False
+ """Whether `check_syntax()` is implemented for this language."""
+
@classmethod
def compile(cls, tmpdir: Path, fnames: list[Path]) -> "CompileResult":
"""Compile, typecheck, and prepare all snippets listed in fnames.
@@ -25,6 +28,17 @@ class Language:
"""
raise NotImplementedError
+ @classmethod
+ def check_syntax(cls, code: str) -> None:
+ """Parse (but do not compile/typecheck) code, raising an exception
+ if it is not syntactically valid.
+
+ Unlike `compile()`, this does not need to resolve imports/types or
+ build a scratch project, so it can run directly on the shortened
+ code extracted from generated markdown files.
+ """
+ raise NotImplementedError
+
@classmethod
def shorten(cls, contents: str) -> dict[str, str]:
"""Shorten the snippet contents into a form suitable for inclusion in
diff --git a/automation/snippets/lib/languages/bash.py b/automation/snippets/lib/languages/bash.py
index 267f74c4a..96eeb558c 100644
--- a/automation/snippets/lib/languages/bash.py
+++ b/automation/snippets/lib/languages/bash.py
@@ -1,3 +1,4 @@
+import subprocess
from pathlib import Path
from .base import CompileResult, Language, copy_template
@@ -6,6 +7,11 @@ from .base import CompileResult, Language, copy_template
class LanguageBash(Language):
NAME = "bash"
SNIPPET_FILENAME = "bash.sh"
+ SUPPORTS_SYNTAX_CHECK = True
+
+ @classmethod
+ def check_syntax(cls, code: str) -> None:
+ subprocess.run(["bash", "-n"], input=code, text=True, check=True)
@classmethod
def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult:
diff --git a/automation/snippets/lib/languages/go.py b/automation/snippets/lib/languages/go.py
index 7e99cc2c9..07d97168b 100644
--- a/automation/snippets/lib/languages/go.py
+++ b/automation/snippets/lib/languages/go.py
@@ -13,10 +13,50 @@ from .base import (
trim_commonpath,
)
+_RE_HEADER_LINE = re.compile(r"^(import|type|func)\b")
+
+
+def _split_header_body(contents: str) -> tuple[str, str]:
+ """Split shortened Go code into leading package-level declarations
+ (imports, types, helper funcs) and the remaining statements that belong
+ inside `func Main()`.
+
+ `shorten()`/`generic_shorten()` flatten these together and dedent
+ everything to column 0, so indentation alone can no longer tell them
+ apart. Helper declarations always come first (`shorten()`'s `RE_CODE`
+ requires `func Main()` to be the last top-level declaration), so this
+ scans for the first line, outside of any brackets, that isn't the start
+ of an `import`/`type`/`func` declaration and treats everything from
+ there onward as the body.
+ """
+ lines = contents.splitlines(keepends=True)
+ depth = 0
+ header_end = 0
+ for i, line in enumerate(lines):
+ if depth == 0:
+ stripped = line.strip()
+ if stripped and _RE_HEADER_LINE.match(stripped) is None:
+ break
+ depth += line.count("{") + line.count("(")
+ depth -= line.count("}") + line.count(")")
+ header_end = i + 1
+ return "".join(lines[:header_end]), "".join(lines[header_end:])
+
class LanguageGo(Language):
NAME = "go"
SNIPPET_FILENAME = "go.go"
+ SUPPORTS_SYNTAX_CHECK = True
+
+ @classmethod
+ def check_syntax(cls, code: str) -> None:
+ subprocess.run(
+ ["gofmt", "-e"],
+ input=cls.unshorten(code),
+ text=True,
+ check=True,
+ stdout=subprocess.DEVNULL,
+ )
@classmethod
def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult:
@@ -94,35 +134,20 @@ class LanguageGo(Language):
def format(cls, fnames: list[str]) -> None:
subprocess.run(["gofmt", "-w", *fnames], check=True)
- RE_RENDERED = re.compile(
- r"""
- (?P
- (?: import\s*\([^)]+\)\n
- | import\s+"[^"]+"\n
- | \n
- )*
- )
- (?P .* )
- $
- """,
- re.DOTALL | re.VERBOSE,
- )
-
@classmethod
def unshorten(cls, contents: str) -> str:
- if m := LanguageGo.RE_RENDERED.match(contents):
- return textwrap.dedent(
- """\
- package snippet
+ header, body = _split_header_body(contents)
+ return textwrap.dedent(
+ """\
+ package snippet
- {imports}
+ {header}
- func Main() {{
- {body}
- }}
- """
- ).format(
- imports=m["imports"].strip(),
- body=textwrap.indent(m["body"].strip(), "\t"),
- )
- return contents
+ func Main() {{
+ {body}
+ }}
+ """
+ ).format(
+ header=header.strip(),
+ body=textwrap.indent(body.strip(), "\t"),
+ )
diff --git a/automation/snippets/lib/languages/python.py b/automation/snippets/lib/languages/python.py
index 5c7ebb12b..90425d1c0 100644
--- a/automation/snippets/lib/languages/python.py
+++ b/automation/snippets/lib/languages/python.py
@@ -1,3 +1,4 @@
+import ast
import re
import subprocess
from pathlib import Path
@@ -19,6 +20,11 @@ RE_IMPORTS = re.compile(
class LanguagePython(Language):
NAME = "python"
SNIPPET_FILENAME = "python.py"
+ SUPPORTS_SYNTAX_CHECK = True
+
+ @classmethod
+ def check_syntax(cls, code: str) -> None:
+ ast.parse(code)
@classmethod
def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult:
diff --git a/automation/snippets/lib/languages/rust.py b/automation/snippets/lib/languages/rust.py
index a678455b1..a7f4d9c41 100644
--- a/automation/snippets/lib/languages/rust.py
+++ b/automation/snippets/lib/languages/rust.py
@@ -18,6 +18,17 @@ from .base import (
class LanguageRust(Language):
NAME = "rust"
SNIPPET_FILENAME = "rust.rs"
+ SUPPORTS_SYNTAX_CHECK = True
+
+ @classmethod
+ def check_syntax(cls, code: str) -> None:
+ subprocess.run(
+ ["rustfmt", "--edition=2024"],
+ input=cls.unshorten(code),
+ text=True,
+ check=True,
+ stdout=subprocess.DEVNULL,
+ )
@classmethod
def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult:
diff --git a/automation/snippets/templates/go/go.mod b/automation/snippets/templates/go/go.mod
index 634db0193..f38f0035c 100644
--- a/automation/snippets/templates/go/go.mod
+++ b/automation/snippets/templates/go/go.mod
@@ -4,14 +4,14 @@ go 1.25.2
require (
github.com/google/uuid v1.6.0
- github.com/qdrant/go-client v1.18.1
+ github.com/qdrant/go-client v1.19.0
)
require (
- golang.org/x/net v0.53.0 // indirect
- golang.org/x/sys v0.43.0 // indirect
- golang.org/x/text v0.36.0 // indirect
+ golang.org/x/net v0.55.0 // indirect
+ golang.org/x/sys v0.45.0 // indirect
+ golang.org/x/text v0.37.0 // indirect
google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4 // indirect
- google.golang.org/grpc v1.80.0 // indirect
+ google.golang.org/grpc v1.82.1 // indirect
google.golang.org/protobuf v1.36.11 // indirect
)
diff --git a/automation/snippets/templates/go/go.sum b/automation/snippets/templates/go/go.sum
index 0387e7a98..2e88c0a22 100644
--- a/automation/snippets/templates/go/go.sum
+++ b/automation/snippets/templates/go/go.sum
@@ -10,31 +10,31 @@ github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
-github.com/qdrant/go-client v1.18.1 h1:o/dDmSl6ONAlaAFtjdlzztcs3NH0tJY3l5C/z/Uu0bE=
-github.com/qdrant/go-client v1.18.1/go.mod h1:Xkfp+r89uNOgSbvilVAhCZ3wKI4G+hB/r9Zr2m4zifI=
+github.com/qdrant/go-client v1.19.0 h1:WGyC1YDXXU8/Od8+kcFncgD3eMqulAoNHy8Ip76tBE8=
+github.com/qdrant/go-client v1.19.0/go.mod h1:/pjMgiL4SxFxD5rLorlS5/FXcvBSx/P2M81gnqn5fb8=
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
go.opentelemetry.io/otel v1.43.0 h1:mYIM03dnh5zfN7HautFE4ieIig9amkNANT+xcVxAj9I=
go.opentelemetry.io/otel v1.43.0/go.mod h1:JuG+u74mvjvcm8vj8pI5XiHy1zDeoCS2LB1spIq7Ay0=
go.opentelemetry.io/otel/metric v1.43.0 h1:d7638QeInOnuwOONPp4JAOGfbCEpYb+K6DVWvdxGzgM=
go.opentelemetry.io/otel/metric v1.43.0/go.mod h1:RDnPtIxvqlgO8GRW18W6Z/4P462ldprJtfxHxyKd2PY=
-go.opentelemetry.io/otel/sdk v1.39.0 h1:nMLYcjVsvdui1B/4FRkwjzoRVsMK8uL/cj0OyhKzt18=
-go.opentelemetry.io/otel/sdk v1.39.0/go.mod h1:vDojkC4/jsTJsE+kh+LXYQlbL8CgrEcwmt1ENZszdJE=
-go.opentelemetry.io/otel/sdk/metric v1.39.0 h1:cXMVVFVgsIf2YL6QkRF4Urbr/aMInf+2WKg+sEJTtB8=
-go.opentelemetry.io/otel/sdk/metric v1.39.0/go.mod h1:xq9HEVH7qeX69/JnwEfp6fVq5wosJsY1mt4lLfYdVew=
+go.opentelemetry.io/otel/sdk v1.43.0 h1:pi5mE86i5rTeLXqoF/hhiBtUNcrAGHLKQdhg4h4V9Dg=
+go.opentelemetry.io/otel/sdk v1.43.0/go.mod h1:P+IkVU3iWukmiit/Yf9AWvpyRDlUeBaRg6Y+C58QHzg=
+go.opentelemetry.io/otel/sdk/metric v1.43.0 h1:S88dyqXjJkuBNLeMcVPRFXpRw2fuwdvfCGLEo89fDkw=
+go.opentelemetry.io/otel/sdk/metric v1.43.0/go.mod h1:C/RJtwSEJ5hzTiUz5pXF1kILHStzb9zFlIEe85bhj6A=
go.opentelemetry.io/otel/trace v1.43.0 h1:BkNrHpup+4k4w+ZZ86CZoHHEkohws8AY+WTX09nk+3A=
go.opentelemetry.io/otel/trace v1.43.0/go.mod h1:/QJhyVBUUswCphDVxq+8mld+AvhXZLhe+8WVFxiFff0=
-golang.org/x/net v0.53.0 h1:d+qAbo5L0orcWAr0a9JweQpjXF19LMXJE8Ey7hwOdUA=
-golang.org/x/net v0.53.0/go.mod h1:JvMuJH7rrdiCfbeHoo3fCQU24Lf5JJwT9W3sJFulfgs=
-golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI=
-golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
-golang.org/x/text v0.36.0 h1:JfKh3XmcRPqZPKevfXVpI1wXPTqbkE5f7JA92a55Yxg=
-golang.org/x/text v0.36.0/go.mod h1:NIdBknypM8iqVmPiuco0Dh6P5Jcdk8lJL0CUebqK164=
+golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8=
+golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww=
+golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
+golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
+golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc=
+golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38=
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4 h1:tEkOQcXgF6dH1G+MVKZrfpYvozGrzb91k6ha7jireSM=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
-google.golang.org/grpc v1.80.0 h1:Xr6m2WmWZLETvUNvIUmeD5OAagMw3FiKmMlTdViWsHM=
-google.golang.org/grpc v1.80.0/go.mod h1:ho/dLnxwi3EDJA4Zghp7k2Ec1+c2jqup0bFkw07bwF4=
+google.golang.org/grpc v1.82.1 h1:NnAxzGRA0677vCa4BUkOAnO5+FfQqVl9iUXeD0IqcGE=
+google.golang.org/grpc v1.82.1/go.mod h1:yzTZ1TB1Z3SG+LIYaI+WiE8D5+PZ3ArnrSp8zF3+/ZA=
google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE=
google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
diff --git a/automation/snippets/templates/python/pyproject.toml b/automation/snippets/templates/python/pyproject.toml
index a44f1c81e..1a2147f89 100644
--- a/automation/snippets/templates/python/pyproject.toml
+++ b/automation/snippets/templates/python/pyproject.toml
@@ -6,11 +6,11 @@ dependencies = [
"datasets>=4.4.1",
"fastembed",
"qdrant-client",
- "qdrant-edge-py==0.7.2"
+ "qdrant-edge-py==0.8.0"
]
[tool.uv.sources]
-qdrant-client = { git = "https://github.com/qdrant/qdrant-client", tag = "v1.18.0" }
+qdrant-client = { git = "https://github.com/qdrant/qdrant-client", tag = "v1.19.0" }
[dependency-groups]
dev = [
diff --git a/automation/snippets/templates/python/uv.lock b/automation/snippets/templates/python/uv.lock
index 688c7bfc8..4c5daf07a 100644
--- a/automation/snippets/templates/python/uv.lock
+++ b/automation/snippets/templates/python/uv.lock
@@ -18,7 +18,7 @@ wheels = [
[[package]]
name = "aiohttp"
-version = "3.14.1"
+version = "3.14.3"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "aiohappyeyeballs" },
@@ -30,90 +30,90 @@ dependencies = [
{ name = "typing-extensions", marker = "python_full_version < '3.13'" },
{ name = "yarl" },
]
-sdist = { url = "https://files.pythonhosted.org/packages/82/78/8ea7308cac6934de8c74a14f3d5f65d1c89287426688be79538d0e5c013d/aiohttp-3.14.1.tar.gz", hash = "sha256:307f2cff90a764d329e77040603fa032db89c5c24fdad50c4c15334cba744035", size = 7955794, upload-time = "2026-06-07T21:09:35.529Z" }
+sdist = { url = "https://files.pythonhosted.org/packages/58/d9/22ce5786ac0c1653ae8b6c23bded02c1686d11f0dbb45b31ce128e0df985/aiohttp-3.14.3.tar.gz", hash = "sha256:9491196535a88924a60afd5b5f434b5b203b6cc616250878dbdb223a8f7844bc", size = 7971213, upload-time = "2026-07-23T01:57:27.037Z" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/1d/21/151624b51cd92553d95424daf4bf19f19ce9be9002d19253e7e7ce67197b/aiohttp-3.14.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:d35143e27778b4bb0fb189562d7f275bff79c62ab8e98459717c0ea617ff2480", size = 757402, upload-time = "2026-06-07T21:06:40.311Z" },
- { url = "https://files.pythonhosted.org/packages/c2/82/280619e0bd7bf2454987e19282616e84762255dd9c8468f62382e8c191f1/aiohttp-3.14.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:bcfb80a2cc36fba2534e5e5b5264dc7ae6fcd9bf15256da3e53d2f499e6fa29d", size = 512310, upload-time = "2026-06-07T21:06:42.207Z" },
- { url = "https://files.pythonhosted.org/packages/55/b2/2aac325583aaa1353045f96dffa586d8a34e8322e14a7ba49cffeb103ab4/aiohttp-3.14.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:27fd7c91e51729b4f7e1577865fa6d34c9adccbc39aabe9000285b48af9f0ec2", size = 512448, upload-time = "2026-06-07T21:06:43.813Z" },
- { url = "https://files.pythonhosted.org/packages/8a/72/a60607cb849faa8af8a356c9329ea2eb6f395d49e82cc82ccba1fd8deb8f/aiohttp-3.14.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:64c567bf9eaf664280116a8688f63016e6b32db2505908e2bdaca1b6438142f2", size = 1766854, upload-time = "2026-06-07T21:06:45.391Z" },
- { url = "https://files.pythonhosted.org/packages/b5/d3/d9fe1c9ec7557ab4d0d82bebaa728c6418f0b93295ec2f4ab015f7710cc7/aiohttp-3.14.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:f5e6ff2bdbb8f4cd3fbe41f99e25bbcd58e3bf9f13d3dd31a11e7917251cc77a", size = 1740884, upload-time = "2026-06-07T21:06:47.413Z" },
- { url = "https://files.pythonhosted.org/packages/c1/dc/f2cecfaf9337ba3e63f181500814ff502aa3d00d9c7ec93a9d23d10a27b2/aiohttp-3.14.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:2f73e01dc37122325caf079982621262f96d74823c179038a82fddfc50359264", size = 1810034, upload-time = "2026-06-07T21:06:50.165Z" },
- { url = "https://files.pythonhosted.org/packages/66/d7/2ff65c5e65c0d7476daf7e15c032e0805e36811185b9623e3238ad6c763e/aiohttp-3.14.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:bb2c0c80d431c0d03f2c7dbf125150fedd4f0de17366a7ca33f7ccb822391842", size = 1904054, upload-time = "2026-06-07T21:06:52.035Z" },
- { url = "https://files.pythonhosted.org/packages/20/9c/d445818389df371f56d141d881153ba23183c4735a03f7356ffb43f7757d/aiohttp-3.14.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3e6fc1a85fa7194a1a7d19f44e8609180f4a8eb5fa4c7ed8b4355f080fad235c", size = 1790278, upload-time = "2026-06-07T21:06:54.049Z" },
- { url = "https://files.pythonhosted.org/packages/4d/aa/bf04cb4d865fc6101c2229a294ad744973b72e513fdc5a6b791e6983d72a/aiohttp-3.14.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:686b6c0d3911ec387b444ddf5dc62fb7f7c0a7d5186a7861626496a5ab4aff95", size = 1591795, upload-time = "2026-06-07T21:06:55.911Z" },
- { url = "https://files.pythonhosted.org/packages/dc/b4/4dac0038960427ba832f6609dfb4ea5437d7fd80c72001b9e48f834f428b/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:c6fa4dc7ad6f8109c70bb1499e589f76b0b792baf39f9b017eb92c8a81d0a199", size = 1728397, upload-time = "2026-06-07T21:06:57.777Z" },
- { url = "https://files.pythonhosted.org/packages/2b/f9/7cd4e8ad7aa3b75f17d56bb5498dd604a93d4e6eece822ba0568c413fff0/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:87a5eea1b2a5e21e1ebdbb33ad4165359189327e63fc4e4894693e7f821ac817", size = 1766504, upload-time = "2026-06-07T21:07:00.009Z" },
- { url = "https://files.pythonhosted.org/packages/f9/df/fc01d9fcad0f73fed3f3d361f1f94f975947b50dff82919f6dc2bf4316cc/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:1c1421eb01d4fd608d88cc8290211d177a58532b55ad94076fb349c5bf467f0a", size = 1777806, upload-time = "2026-06-07T21:07:02.064Z" },
- { url = "https://files.pythonhosted.org/packages/41/09/47e2d090bddcc8fb4ccb4c314aadc32d7c5d9bb55f50f6ad1c92fc15d501/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:34b257ec41345c1e8f2df68fa908a7952f5de932723871eb633ecbbff396c9a4", size = 1580707, upload-time = "2026-06-07T21:07:03.942Z" },
- { url = "https://files.pythonhosted.org/packages/3d/36/f1a4ce904ae0b6930cfe9afc96d0896f7ec1a620c400405d63783bb95a9c/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:de538791a80e5d862addbc183f70f0158ac9b9bb872bb147f1fd2a683691e087", size = 1798121, upload-time = "2026-06-07T21:07:05.987Z" },
- { url = "https://files.pythonhosted.org/packages/70/0a/e0075ce9ca0279ee1d4f0c0b85f54fea02ebc83c3007651a72bece658fec/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:6f71173be42d3241d428f760122febb748de0623f44308a6f120d0dd9ec572e3", size = 1767580, upload-time = "2026-06-07T21:07:07.873Z" },
- { url = "https://files.pythonhosted.org/packages/3e/61/a0c0a8f327a9c52095cdd8e312391b00d3ed64ab6c72bb5c33d8ec251cf7/aiohttp-3.14.1-cp312-cp312-win32.whl", hash = "sha256:ec8dc383ee57ea3e883477dcca3f11b65d58199f1080acaf4cd6ad9a99698be4", size = 452771, upload-time = "2026-06-07T21:07:09.669Z" },
- { url = "https://files.pythonhosted.org/packages/df/d9/ea367c75f16ac9c6cdc8febb25e8318fa21a2b1bc8d6514d4b2d890bface/aiohttp-3.14.1-cp312-cp312-win_amd64.whl", hash = "sha256:2aa92c87868cd13674989f9ee83e5f9f7ea4237589b728048e1f0c8f6caa3271", size = 479873, upload-time = "2026-06-07T21:07:11.538Z" },
- { url = "https://files.pythonhosted.org/packages/03/64/8d96784a7851156db8a4c6c3f6f91042fdf39fb15a4cc38c8b3c14833c45/aiohttp-3.14.1-cp312-cp312-win_arm64.whl", hash = "sha256:2c840c90759922cb5e6dda94596e079a30fb5a5ba548e7e0dc00574703940847", size = 448073, upload-time = "2026-06-07T21:07:13.637Z" },
- { url = "https://files.pythonhosted.org/packages/bc/97/bd137012dd97e1649162b099135a80e1fd59aaa807b2430fc448d1029aff/aiohttp-3.14.1-cp313-cp313-android_21_arm64_v8a.whl", hash = "sha256:b3a03285a7f9c7b016324574a6d92a1c895da6b978cb8f1deee3ac72bc6da178", size = 506882, upload-time = "2026-06-07T21:07:15.501Z" },
- { url = "https://files.pythonhosted.org/packages/ef/79/e5cc690e9d922a66887ceeaca53a8ffd5a7b0be3816142b7abc433742d89/aiohttp-3.14.1-cp313-cp313-android_21_x86_64.whl", hash = "sha256:2a73f487ab8ef5abbb24b7aa9b73e98eaba9e9e031804ff2416f02eca315ccaf", size = 515270, upload-time = "2026-06-07T21:07:17.53Z" },
- { url = "https://files.pythonhosted.org/packages/fe/22/a73ccbf9dbd6e26dda0b24d5fd5db7da92ee3383a79f47677ffb834c5c5b/aiohttp-3.14.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:915fbb7b41b115192259f8c9ae58f3ddc444d2b5579917270211858e606a4afd", size = 485841, upload-time = "2026-06-07T21:07:19.555Z" },
- { url = "https://files.pythonhosted.org/packages/3b/b9/57ed8eaf596321c2ad747bd480fb1700dbd7177c60dfc9e4c187f629662e/aiohttp-3.14.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:7fb4bdf95b0561a79f259f9d28fbc109728c5ee7f27aff6391f0ca703a329abe", size = 492088, upload-time = "2026-06-07T21:07:21.581Z" },
- { url = "https://files.pythonhosted.org/packages/78/c0/5ebe5270a7c140d7c6f79dcb018640225f14d406c149e4eec04a7d82fe71/aiohttp-3.14.1-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:1b9748363260121d2927704f5d4fc498150669ca3ae93625986ee89c8f80dcd4", size = 501564, upload-time = "2026-06-07T21:07:23.388Z" },
- { url = "https://files.pythonhosted.org/packages/75/7f/8cdaa24fc7983865e0915153b96a9ac5bcdd3548d64c5a27d17cecccad2d/aiohttp-3.14.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:86a6dab78b0e43e2897a3bbe15745aa60dc5423ca437b7b0b164c069bf91b876", size = 751998, upload-time = "2026-06-07T21:07:25.046Z" },
- { url = "https://files.pythonhosted.org/packages/b2/f4/c4227aacfacc5cb0cc2d119b65301d177912a6842cd64e120c47af76064f/aiohttp-3.14.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:4dfd6e47d3c44c2279907607f73a4240b88c69eb8b90da7e2441a8045dfd21da", size = 510918, upload-time = "2026-06-07T21:07:27.28Z" },
- { url = "https://files.pythonhosted.org/packages/ab/01/a2d5f96cd4e74424864d30bc0a7e44d0a12dacdcfa91b5b2d1bd3dca6bf3/aiohttp-3.14.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:317acd9f8602858dc7d59679812c376c7f0b97bcbbf16e0d6237f54141d8a8a6", size = 508657, upload-time = "2026-06-07T21:07:29.252Z" },
- { url = "https://files.pythonhosted.org/packages/e8/ed/3c0fb5c500fdd8e7ebc10d1889c04384fffa1a9163eac1356088ca9da1b1/aiohttp-3.14.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bd869c427324e5cb15195793de951295710db28be7d818247f3097b4ab5d4b96", size = 1757907, upload-time = "2026-06-07T21:07:31.03Z" },
- { url = "https://files.pythonhosted.org/packages/0b/ab/d4c924d9bd5be3050c226612413ce68cb54c70d2c31b661bfc8d9a5b6a70/aiohttp-3.14.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:93b032b5ec3255473c143627d21a69ac74ae12f7f33974cb587c564d11b1066f", size = 1737565, upload-time = "2026-06-07T21:07:33.031Z" },
- { url = "https://files.pythonhosted.org/packages/19/2a/37326821ff779084020cdc33224d20b19f42f4183a500ff92022a739eda7/aiohttp-3.14.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f234b4deb12f3ad59127e037bc57c40c21e45b45282df7d3a55a0f409f595296", size = 1799018, upload-time = "2026-06-07T21:07:35.003Z" },
- { url = "https://files.pythonhosted.org/packages/b3/4f/6e947ba73e4ce09070761c05ed3a8ceb7c21f5e46798671d8b2aac0e4626/aiohttp-3.14.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:9af6779bfb46abf124068327abcdf9ce95c9ef8287a3e8da76ccf2d0f16c28fa", size = 1894416, upload-time = "2026-06-07T21:07:36.956Z" },
- { url = "https://files.pythonhosted.org/packages/9d/6e/dbf1d0625dc711fb2851f4f3c3055c39ed58bae92082d8c627dbe6013736/aiohttp-3.14.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:faccab372e66bc76d5731525e7f1143c922271725b9d38c9f97edcc66266b451", size = 1783881, upload-time = "2026-06-07T21:07:39.063Z" },
- { url = "https://files.pythonhosted.org/packages/44/c2/5e25098a67268ed369483ae7d1a58bd0a13d03aab860d2a0e4a6eb25b046/aiohttp-3.14.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f380468b09d2a81633ee863b0ec5648d364bd17bb8ecfb8c2f387f7ac1faf42c", size = 1587572, upload-time = "2026-06-07T21:07:41.058Z" },
- { url = "https://files.pythonhosted.org/packages/2a/bd/cf9cee17e140f942a3de73e658a543aa8fbf35a5fc67a9d2538d52d77f0b/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:97e704dcd26271f5bda3fa07c3ce0fb76d6d3f8659f4baa1a24442cc9ba177ca", size = 1722137, upload-time = "2026-06-07T21:07:43.014Z" },
- { url = "https://files.pythonhosted.org/packages/89/6d/5684f8c59045c96f81a18cefbc1fbbd79d25b88f1c622f2a5c5c08fcb632/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:269b76ac5394092b95bc4a098f4fc6c191c083c3bd12775d1e30e663132f6a09", size = 1755953, upload-time = "2026-06-07T21:07:45.933Z" },
- { url = "https://files.pythonhosted.org/packages/a8/40/35caf3170f8359760740a7d9aa0fff2e344bef98e1d1186f5a0f6dec17e6/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:5c0b3e614340c889d575451696374c9d17affd54cd607ca0babed8f8c37b9397", size = 1766479, upload-time = "2026-06-07T21:07:48.047Z" },
- { url = "https://files.pythonhosted.org/packages/6d/a1/b0c61e7a137f0d81de49a82023a6df73c3c16d6fefb0f8e4a93d21639002/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:5663ee9257cfa1add7253a7da3035a02f31b6600ec48261585e1800a81533080", size = 1580077, upload-time = "2026-06-07T21:07:50.069Z" },
- { url = "https://files.pythonhosted.org/packages/0b/41/194ea4623693009fcefebef7aef63c141754f153e9cd0d39d3b9e36c175c/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:603a2c834142172ffddc054067f5ec0ca65d57a0aa98a71bc81952573208e345", size = 1791688, upload-time = "2026-06-07T21:07:52.106Z" },
- { url = "https://files.pythonhosted.org/packages/ba/45/4de841f005cfe1fd63e2a2fe011262c515e2a62aa6994b15947e7d717ac9/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:cb21957bb8aca671c1765e32f58164cf0c50e6bf41c0bbbd16da20732ecaf588", size = 1761094, upload-time = "2026-06-07T21:07:54.113Z" },
- { url = "https://files.pythonhosted.org/packages/e4/ae/dbce10533d3896d544d5053939ed75b7dc31a1b0973d959b1b5ae21028d6/aiohttp-3.14.1-cp313-cp313-win32.whl", hash = "sha256:e509a55f681e6158c20f70f102f9cf61fb20fbc382272bc6d94b7343f2582780", size = 452662, upload-time = "2026-06-07T21:07:56.06Z" },
- { url = "https://files.pythonhosted.org/packages/7b/d9/0bf1a19362c32f06229da5e7ddfcec91f93474d6307f7a2d3135e9c674dc/aiohttp-3.14.1-cp313-cp313-win_amd64.whl", hash = "sha256:1ac8531b638959718e18c2207fbfe297819875da46a740b29dfa29beba64355a", size = 479748, upload-time = "2026-06-07T21:07:58.319Z" },
- { url = "https://files.pythonhosted.org/packages/22/0a/62e7232dc9484fbec112ceb32efb6a624cc7994ec6e2b019286f17c4e8f2/aiohttp-3.14.1-cp313-cp313-win_arm64.whl", hash = "sha256:250d14af67f6b6a1a4a811049b1afa69d61d617fca6bf33149b3ab1a6dbcf7b8", size = 447723, upload-time = "2026-06-07T21:08:00.154Z" },
- { url = "https://files.pythonhosted.org/packages/c4/a1/5fafa04e1ca91ddb47608699d60649c1c6db3cf41c99e78fc4056f9513db/aiohttp-3.14.1-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:7c106c26852ca1c2047c6b80384f17100b4e439af276f21ef3d4e2f450ae7e15", size = 508531, upload-time = "2026-06-07T21:08:02.093Z" },
- { url = "https://files.pythonhosted.org/packages/fa/2e/bfa02f699d87ffc86d5959270b28f1cb410add3ccaced8ed2e0b8a5238fc/aiohttp-3.14.1-cp314-cp314-android_24_x86_64.whl", hash = "sha256:20205f7f5ade7aaec9f4b500549bbc071b046453aed72f9c06dcab87896a83e8", size = 514718, upload-time = "2026-06-07T21:08:04.476Z" },
- { url = "https://files.pythonhosted.org/packages/85/a5/9594ad6289eebbc97d167c44213d557807f90e59115caad24de21ad2c3b1/aiohttp-3.14.1-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:62a759436b29e677181a9e76bab8b8f689a29cb9c535f45f7c48c9c830d3f8c3", size = 487918, upload-time = "2026-06-07T21:08:06.377Z" },
- { url = "https://files.pythonhosted.org/packages/b4/61/16a32c36c3c49edec122a3dc811f2057df2f94d3b14aa107c8017d981618/aiohttp-3.14.1-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:2964cbf553df4d7a57348da44d961d871895fc1ee4e8c322b2a95612c7b17fba", size = 494014, upload-time = "2026-06-07T21:08:08.263Z" },
- { url = "https://files.pythonhosted.org/packages/9b/89/3ebcf96ed99c05bec9c434aaac6963fd3cbab4a786ae739908a144d9ce44/aiohttp-3.14.1-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:237651caadc3a59badd39319c54642b5299e9cc98a3a194310e55d5bb9f5e397", size = 502398, upload-time = "2026-06-07T21:08:10.244Z" },
- { url = "https://files.pythonhosted.org/packages/fd/3d/b74870a0c2d40c355928cd5b96c7a11fa821b8a40fc41365e64479b151fb/aiohttp-3.14.1-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:896e12dfdbbab9d8f7e16d2b28c6769a60126fa92095d1ebf9473d02593a2448", size = 758018, upload-time = "2026-06-07T21:08:12.447Z" },
- { url = "https://files.pythonhosted.org/packages/d3/66/f42f5c984d99e49c6cff5f26f590750f2e2f7ef1fcfb99966ab5be1b632e/aiohttp-3.14.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:d03f281ed22579314ba00821ce20115a7c0ac430660b4cc05704a3f818b3e004", size = 512462, upload-time = "2026-06-07T21:08:14.624Z" },
- { url = "https://files.pythonhosted.org/packages/e9/a7/248e1aebe0c7810b0271e021a0f2a5eb6e78a051885b3c9df49f42a5802d/aiohttp-3.14.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:07eabb979d236335fed927e137a928c9adfb7df3b9ec7aa31726f133a62be983", size = 512824, upload-time = "2026-06-07T21:08:16.572Z" },
- { url = "https://files.pythonhosted.org/packages/26/97/2aa0e5ba0727dc3bd5aaebb7ccbc510f7dfb7fb961ec87497cd496635ab1/aiohttp-3.14.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4fe1f1087cbadb280b5e1bb054a4f00d1423c74d6626c5e48400d871d34ecefe", size = 1749898, upload-time = "2026-06-07T21:08:18.635Z" },
- { url = "https://files.pythonhosted.org/packages/00/8d/e97f6c96c891d457c8479d92a514ba194d0412f981d72c70341ee18488ed/aiohttp-3.14.1-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:367a9314fdc79dab0fac96e216cb41dd73c85bdca85306ce8999118ba7e0f333", size = 1710114, upload-time = "2026-06-07T21:08:20.892Z" },
- { url = "https://files.pythonhosted.org/packages/6f/e6/aa8d7e863048c8fceb5cd6ce74017311cec3ead07847387e12265fb4444e/aiohttp-3.14.1-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a24f677ebe83749039e7bdf862ff0bbb16818ae4193d4ef96505e269375bcce0", size = 1802541, upload-time = "2026-06-07T21:08:23.044Z" },
- { url = "https://files.pythonhosted.org/packages/83/a8/72193137de57fda4ebfae4563182d082c8856e3b6e9871d0b46f028fb369/aiohttp-3.14.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c83afe0ba876be7e943d2e0ba645809ad441575d2840c895c21ee5de93b9377a", size = 1875776, upload-time = "2026-06-07T21:08:25.288Z" },
- { url = "https://files.pythonhosted.org/packages/a0/18/938441025db6769a3464596b2410af3afde0b21eb2f204c6f766f68af4bd/aiohttp-3.14.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:634e385930fb6d2d479cf3aa66515955863b77a5e3c2b5894ca259a25b308602", size = 1760329, upload-time = "2026-06-07T21:08:27.363Z" },
- { url = "https://files.pythonhosted.org/packages/60/29/bf2496b4065e76e09fe48015aaffe5ce161d8f089b06ac6982070f653076/aiohttp-3.14.1-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:eeea07c4397bbc57719c4eed8f9c284874d4f175f9b6d57f7a1546b976d455ca", size = 1587293, upload-time = "2026-06-07T21:08:29.805Z" },
- { url = "https://files.pythonhosted.org/packages/49/a2/2136674d52123b1354bd05dd5753c318db47dc0c927cc70b27bab3755456/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:335c0cc3e3545ce98dcb9cfcb836f40c3411f43fa03dab757597d80c89af8a35", size = 1714756, upload-time = "2026-06-07T21:08:32.094Z" },
- { url = "https://files.pythonhosted.org/packages/a7/b9/e5fd2e6f915503081c0f9b1e8540947037929c70c191da2e4d54b31a21a1/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:ae6be797afdef264e8a84864a85b196ca06045586481b3df8a967322fd2fa844", size = 1721052, upload-time = "2026-06-07T21:08:34.167Z" },
- { url = "https://files.pythonhosted.org/packages/63/5a/2833e324a2263e104e31e2e91bc5bbee81bc499afd32203faee048a883f0/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:8560b4d712474335d08907db7973f71912d3a9a8f1dee992ec06b5d2fe359496", size = 1766888, upload-time = "2026-06-07T21:08:36.95Z" },
- { url = "https://files.pythonhosted.org/packages/57/fa/dea6511870913162f3b2e8c42a7614eb203a4540b8c2da43e0bfb0548f3c/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:2b7edd08e0a5deb1e8564a2fcd8f4561014a3f05252334671bbf55ddd47db0e5", size = 1581679, upload-time = "2026-06-07T21:08:39.292Z" },
- { url = "https://files.pythonhosted.org/packages/14/bd/3cf0d55e71784b33534e9710a67d382d900598b4787fbce6cc7317f8c42a/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:b6ff7fcee63287ae57b5df3e4f5957ce032122802509246dec1a5bcc55904c95", size = 1782021, upload-time = "2026-06-07T21:08:41.407Z" },
- { url = "https://files.pythonhosted.org/packages/c1/af/14bb5843eccbe234f4dfb78ab73e549d99727247e62ae5d62cbd22eaf5b0/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:6ffbb2f4ec1ceaff7e07d43922954da26b223d188bf30658e561b98e23089444", size = 1742574, upload-time = "2026-06-07T21:08:43.795Z" },
- { url = "https://files.pythonhosted.org/packages/f2/1e/fbeb7af9210a67ac0f9c9bec0f8f4568497924e33137a3d5b48e1cf85f3f/aiohttp-3.14.1-cp314-cp314-win32.whl", hash = "sha256:a9875b46d910cff3ea2f5962f9d266b465459fe634e22556ab9bd6fc1192eea0", size = 457773, upload-time = "2026-06-07T21:08:46.168Z" },
- { url = "https://files.pythonhosted.org/packages/f0/2b/13e8d741a9ec5db7d900c060554cf8352ab85e44e2a4469ebb9d377bda17/aiohttp-3.14.1-cp314-cp314-win_amd64.whl", hash = "sha256:af8b4b81a960eeaf1234971ac3cd0ba5901f3cd42eae42a46b4d089a8b492719", size = 485001, upload-time = "2026-06-07T21:08:48.401Z" },
- { url = "https://files.pythonhosted.org/packages/df/30/491acfa2c4d6c3ff59c49a14fc1b50be3241e25bbb0c84c09e2da4d11395/aiohttp-3.14.1-cp314-cp314-win_arm64.whl", hash = "sha256:cf4491381b1b57425c315a56a439251b1bdac07b2275f19a8c44bc57744532ec", size = 453809, upload-time = "2026-06-07T21:08:50.7Z" },
- { url = "https://files.pythonhosted.org/packages/34/e3/19dbe1a1f4cc6230eb9e314de7fe68053b0992f9302b27d12141a0b5db53/aiohttp-3.14.1-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:819c054312f1af92947e6a55883d1b66feefab11531a7fc45e0fb9b63880b5c2", size = 793320, upload-time = "2026-06-07T21:08:52.775Z" },
- { url = "https://files.pythonhosted.org/packages/7f/20/1b7182219ba1b108430d6e4dc53d25ae02dcfcf5a045b33af4e8c5167527/aiohttp-3.14.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:10ee9c1753a8f706345b22496c79fbddb5be0599e0823f3738b1534058e25340", size = 529077, upload-time = "2026-06-07T21:08:55Z" },
- { url = "https://files.pythonhosted.org/packages/b9/c8/14ce60ec31a2e5f5274bb17d383a6f7a3aabca31ac04eee05585bbadab16/aiohttp-3.14.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1601cc37baf5750ccacae618ec2daf020769581695550e3b654a911f859c563d", size = 532476, upload-time = "2026-06-07T21:08:57.176Z" },
- { url = "https://files.pythonhosted.org/packages/7e/02/9ac85e081e53da2e061b02fa7758fe0a12d17b8ce2d1f5e6c7cb76730328/aiohttp-3.14.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4d6e0ac9da31c9c04c84e1c0182ad8d6df35965a85cae29cd71d089621b3ae94", size = 1922347, upload-time = "2026-06-07T21:08:59.563Z" },
- { url = "https://files.pythonhosted.org/packages/c0/3e/d3ba07a0ab38b5389e10bec4362d21e10a4f667cba2d79ba30837b3a5059/aiohttp-3.14.1-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:9e8f2d660c350b3d0e259c7a7e3d9b7fc8b41210cbcc3d4a7076ff0a5e5c2fdc", size = 1786465, upload-time = "2026-06-07T21:09:01.909Z" },
- { url = "https://files.pythonhosted.org/packages/0b/cb/e2ee978a00cfb2df829704a69528b18154eba5939f45bc1efa8f33aee4c5/aiohttp-3.14.1-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4691802dda97be727f79d86818acaad7eb8e9252626a1d6b519fedbb92d5e251", size = 1909423, upload-time = "2026-06-07T21:09:04.357Z" },
- { url = "https://files.pythonhosted.org/packages/73/5d/1430334858b1022b58ae50399a918f0bd6fe8fa7fa183598d657ff61e040/aiohttp-3.14.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c389c482a7e9b9dc3ee2701ac46c4125297a3818875b9c305ddb603c04828fd1", size = 2001906, upload-time = "2026-06-07T21:09:06.722Z" },
- { url = "https://files.pythonhosted.org/packages/66/4e/560c7472d3d198a23aa5c8b19a5115bf6a9b77b7d3e4bb363da320430ad2/aiohttp-3.14.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fc0cacab7ba4e56f0f81c82a98c09bed2f39c940107b03a34b168bdf7597edd3", size = 1877095, upload-time = "2026-06-07T21:09:09.011Z" },
- { url = "https://files.pythonhosted.org/packages/0d/f1/4745806578d447db4a784a8591e2dae3afdfc2bcb96f8f81271b13df6543/aiohttp-3.14.1-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:979ed4717f59b8bb12e3963378fa285d93d367e15bcd66c721311826d3c44a6c", size = 1676222, upload-time = "2026-06-07T21:09:11.461Z" },
- { url = "https://files.pythonhosted.org/packages/6a/c9/48255813cca749a229ef0ab476004ec623728ad79a9c0840616f6c076325/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:38e1e7daaea81df51c952e18483f323d878499a1e2bfe564790e0f9701d6f203", size = 1842922, upload-time = "2026-06-07T21:09:14.118Z" },
- { url = "https://files.pythonhosted.org/packages/3d/c0/bbd054e2bee909f529523a5af3891052606af5143c09f5f183ec3b234676/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:4132e72c608fe9fecb8f409113567605915b83e9bdd3ea56538d2f9cd35002f1", size = 1825035, upload-time = "2026-06-07T21:09:16.447Z" },
- { url = "https://files.pythonhosted.org/packages/a8/ae/90395d4376deceb74e09ec26b6adf7d2015a6f8802d6d84446af860fef04/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:eefd9cc9b6d4a2db5f00a26bc3e4f9acf71926a6ec557cd56c9c6f27c290b665", size = 1849512, upload-time = "2026-06-07T21:09:18.742Z" },
- { url = "https://files.pythonhosted.org/packages/93/bd/fb25f3049957553d4ce0ba6ae480aa2f592a6985497fca590837d16c1be0/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:b165790117eea512d7f3fb22f1f6dad3d55a7189571993eb015591c1401276d1", size = 1668571, upload-time = "2026-06-07T21:09:21.458Z" },
- { url = "https://files.pythonhosted.org/packages/3f/22/7f73303d64dd567ff3addca90b556690ed1233a47b8f55d242fb90af3681/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:ed09c7eb1c391271c2ed0314a51903e72a3acb653d5ccfc264cdf3ef11f8269d", size = 1881159, upload-time = "2026-06-07T21:09:23.813Z" },
- { url = "https://files.pythonhosted.org/packages/44/be/0474c5a8b5640e1e4aa1923430a91f4151be82e511373fe764189b89aef5/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:99abd37084b82f5830c635fddd0b4993b9742a66eb746dacf433c8590e8f9e3c", size = 1841409, upload-time = "2026-06-07T21:09:26.207Z" },
- { url = "https://files.pythonhosted.org/packages/7b/3c/bb4a7cba26956cb3da4553cc2056cf67be5b5ff6e6d8fa4fbdff73bfb7ae/aiohttp-3.14.1-cp314-cp314t-win32.whl", hash = "sha256:47ddf841cdecc810749921d25606dee45857d12d2ad5ddb7b5bd7eab12e4b365", size = 494166, upload-time = "2026-06-07T21:09:28.505Z" },
- { url = "https://files.pythonhosted.org/packages/8a/84/ec80c2c1f66a952555a9f86df6b33af65108a6febfa0471b69013a12f807/aiohttp-3.14.1-cp314-cp314t-win_amd64.whl", hash = "sha256:5e78b522b7a6e27e0b25d19b247b75039ac4c94f99823e3c9e53ae1603a9f7e9", size = 530255, upload-time = "2026-06-07T21:09:30.843Z" },
- { url = "https://files.pythonhosted.org/packages/2a/71/6e22be134a4061ada85a92951b842f2657f17d926b727f3f94c56ae963d6/aiohttp-3.14.1-cp314-cp314t-win_arm64.whl", hash = "sha256:90d53f1609c29ccc2193945ef732428382a28f78d0456ae4d3daf0d48b74f0f6", size = 469640, upload-time = "2026-06-07T21:09:33.028Z" },
+ { url = "https://files.pythonhosted.org/packages/18/d4/eb96299230e20acf2efae207cb8d69051f1f68e357e5ea5e479bf6fb097a/aiohttp-3.14.3-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:39aded8c7f3b935b54aab1d8d73c70ec0ee2d3ec3b943e0e86611bc150ba47f5", size = 754690, upload-time = "2026-07-23T01:53:47.332Z" },
+ { url = "https://files.pythonhosted.org/packages/88/11/e7a70a209eb9a067c0d3212b518a0134e3484f5178c7533878b6b514d469/aiohttp-3.14.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:5bcb6ff3fdab1258a192679ff1a05d44f59626430aa05cd1a9d2447423599228", size = 509484, upload-time = "2026-07-23T01:53:51.159Z" },
+ { url = "https://files.pythonhosted.org/packages/30/07/4bbc222cc8dbe31d4c3e8a5baad2286e4d42026ac0c570027b89afce6344/aiohttp-3.14.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:617105e2c3018ee38d0c8ce5ee3c84f621a6d8b9f723202aacaff28449ca91ee", size = 511949, upload-time = "2026-07-23T01:53:55.083Z" },
+ { url = "https://files.pythonhosted.org/packages/54/b9/42e74c46b7b7c794b995bbc1f573fb48950c38b19d8600c62a6804ee2d67/aiohttp-3.14.3-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f631fe87a6f30df5fbe6d79640b25e4cffb38c31c7fb6f10871517b84b0f8c1a", size = 1765282, upload-time = "2026-07-23T01:53:59.662Z" },
+ { url = "https://files.pythonhosted.org/packages/6b/ed/62bc4d74363ad346d518e0720363a949f63e2e23439a79eb5813d4d29bb3/aiohttp-3.14.3-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:a94dbaae5ae27bd849c93570669bff91e0510f33a80805738e3de72a7be0447b", size = 1741511, upload-time = "2026-07-23T01:54:04.063Z" },
+ { url = "https://files.pythonhosted.org/packages/d0/9f/181e8a8bc79e47d13c7fc4540bd7a3b729d9505609c61f392a8dd2fbfe55/aiohttp-3.14.3-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:8f2f1c4c032c7cedd7d8da6f54c97b70266c6570c3108d3fdffee7188bb70529", size = 1810680, upload-time = "2026-07-23T01:54:09.882Z" },
+ { url = "https://files.pythonhosted.org/packages/5c/9a/dec94d6ad694552fe3424e3f1928d7a606a5d9d9433a04e7ecdd9d38ae7f/aiohttp-3.14.3-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ea05e1f97ceea523942d9b2a7d7c0359d781d683d6b043f5943a602b14da4787", size = 1905646, upload-time = "2026-07-23T01:54:13.475Z" },
+ { url = "https://files.pythonhosted.org/packages/52/b7/7cd31f29d6055bd711ae6e669367fba6f5ae9de463910a793e30556a8db7/aiohttp-3.14.3-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:543906c127fb1d929b95076db19b83fa2d46751006ff1e23b093aa5ac4d8db42", size = 1792122, upload-time = "2026-07-23T01:54:15.752Z" },
+ { url = "https://files.pythonhosted.org/packages/66/73/10b1ef93afa61f4963c746257b70ced619cf31a4798671de5fdb2608501d/aiohttp-3.14.3-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0a5ff2dfbb9ce645fa5b8ef3e02c6c0b9cc3f6030ff863d0c51fffc50cb5541b", size = 1591127, upload-time = "2026-07-23T01:54:19.489Z" },
+ { url = "https://files.pythonhosted.org/packages/49/ed/3b203fa6de1b338c14acdc06bf6ca9b043b7944f005966958c2ced932cde/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:041badb8f84396357c4d3ad26de6afd7a32b112f43d3c63045c0c8278cfd2043", size = 1725210, upload-time = "2026-07-23T01:54:24.129Z" },
+ { url = "https://files.pythonhosted.org/packages/28/b7/1c2aab8c706436dcc28598452488ac9cd7c409da815237c28c27d58993e6/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:530125ee1163c4219af35dc3aa1206e541e7b31b6efc1a3f93b70a136f65d427", size = 1764848, upload-time = "2026-07-23T01:54:27.973Z" },
+ { url = "https://files.pythonhosted.org/packages/54/50/94c28f08b131c4bf10984ea2c7a536c9920608bb2d6e7f95642c30cc87b7/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:c8653fd547c93a61aadc612007790f5555cdd18946fa48cf45e26d8ea4ea473d", size = 1777102, upload-time = "2026-07-23T01:54:31.775Z" },
+ { url = "https://files.pythonhosted.org/packages/13/d4/e7d09ba7d345fb2d74440fd2fa033c5e079fac05552927705986f41a364f/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:89176250f686cb9853c0fb7ead90e639e915b84a6f43eedc2a4e7ec21f1037f0", size = 1580205, upload-time = "2026-07-23T01:54:34.518Z" },
+ { url = "https://files.pythonhosted.org/packages/a3/84/072a91d68e1e1eb587985b54baab94221277f877e8ef274fc213a0ceae28/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:3a26434dafe408229ff3403458ca58de24fb51936504decac49ce6755f77e59d", size = 1797219, upload-time = "2026-07-23T01:54:36.995Z" },
+ { url = "https://files.pythonhosted.org/packages/e0/eb/aad34e897e668424d6e995da5dff8a4a09af93363d3392488772957a63aa/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:d1558173930a5a8d3069cee5c92fc91c87c4dbcb099debbb3622053717145a19", size = 1768629, upload-time = "2026-07-23T01:54:40.103Z" },
+ { url = "https://files.pythonhosted.org/packages/b6/2b/6bb88ddba0fecd9122aa3ebcad25996cf6c083a4a7040dbb3a4f97972af6/aiohttp-3.14.3-cp312-cp312-win32.whl", hash = "sha256:16100ad3ab8d649fdfbee87602d9d2dcdca9df0b9eda8a1b5fdc0d41f96da559", size = 451481, upload-time = "2026-07-23T01:54:42.547Z" },
+ { url = "https://files.pythonhosted.org/packages/76/9b/f2f8f108da17ecef2cc3efc424e8b7ad3782b1a8360f7b8eae8ced84f6ea/aiohttp-3.14.3-cp312-cp312-win_amd64.whl", hash = "sha256:33a2d7c28d33797a2e99923dffa63f83d908a19b6bf26cfe80fa790aa5e1a75a", size = 476845, upload-time = "2026-07-23T01:54:44.853Z" },
+ { url = "https://files.pythonhosted.org/packages/3e/44/28dac80a8941b604f4da10ce21097614ca1bf905ce93dca28d8d7de9c1e7/aiohttp-3.14.3-cp312-cp312-win_arm64.whl", hash = "sha256:362a3fd481769cac1a824514bcd86fda51c65e8fe6e051099e008fddde6db17c", size = 448050, upload-time = "2026-07-23T01:54:47.087Z" },
+ { url = "https://files.pythonhosted.org/packages/57/be/5afd201cc0ab139029aadb75392efe85a293403d9dd3a3226161c21ce00c/aiohttp-3.14.3-cp313-cp313-android_21_arm64_v8a.whl", hash = "sha256:2e9878ae68e4a5f1c0abe4dd497dbc3d51946f5837b56759e2a02e78fa90ef86", size = 506269, upload-time = "2026-07-23T01:54:49.075Z" },
+ { url = "https://files.pythonhosted.org/packages/22/09/dec8189d62b45ade009f6792a2264b942a90cb88aeaf181239933cd72c3c/aiohttp-3.14.3-cp313-cp313-android_21_x86_64.whl", hash = "sha256:f3d2669fe7dec7fc359ecdb5984b29b50d85d5d00f8c1cb61de4f4a24ee42627", size = 515166, upload-time = "2026-07-23T01:54:51.894Z" },
+ { url = "https://files.pythonhosted.org/packages/28/24/2854869d29ed8a8b19d74f9ec6629515f7e04d02dd329d9d179201e58e47/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:cc7cb243a68167172f48c1fd43cee91ec4b1d40cefd190edd43369d1a6bc9c82", size = 486263, upload-time = "2026-07-23T01:54:54.223Z" },
+ { url = "https://files.pythonhosted.org/packages/d4/dd/57187c8be2a35aea65eaee3bd2c3dcbbcf0204f5106c89637e3610380cd1/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:78253b573e6ffab5028924fc98bc281aae05445969982a10864bc360dea2016c", size = 492299, upload-time = "2026-07-23T01:54:56.236Z" },
+ { url = "https://files.pythonhosted.org/packages/b9/11/06ae6ed8f0d414edf4068861e233d8fe23ee699bfd4b3ceb8663db948a62/aiohttp-3.14.3-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:7041d52c3a7fa20c9e8c182b534704abb19502c8bdcbde7ab23bfda6f642394f", size = 502235, upload-time = "2026-07-23T01:54:58.377Z" },
+ { url = "https://files.pythonhosted.org/packages/7e/a3/559639c34a345d2cf7c52dff6838119f2eaf29eb508227b5b83f573af813/aiohttp-3.14.3-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:ac74facc01463f138b0da5580329cfcc82818dea5656e83ddcd11268fc12ff80", size = 750883, upload-time = "2026-07-23T01:55:00.65Z" },
+ { url = "https://files.pythonhosted.org/packages/91/cd/41e131f13afd1e7b0172a9d9eda085ef90eb8439f41f0d279db81ed3ae60/aiohttp-3.14.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:d6218d92e450824e9b4881f44e8c09f1853b490f9a64130801024a4793b1b3b0", size = 508473, upload-time = "2026-07-23T01:55:02.945Z" },
+ { url = "https://files.pythonhosted.org/packages/bc/6b/e7f13410d391c6e55b4c007a8de024355389d7d459e3d64c42b2d33617e5/aiohttp-3.14.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:11fb37ef075669eee52ab1928fbf6e1741fada40409fa309ebde9607a962aebf", size = 509190, upload-time = "2026-07-23T01:55:05.173Z" },
+ { url = "https://files.pythonhosted.org/packages/97/21/6464573e53d69672cc1eada3e5c5cb2d2efa82701e8305a0f2047a576967/aiohttp-3.14.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:55bdcc472aafe2de4a253045cc128007a64f1e0264fb675791e132ea5edaa3bd", size = 1761478, upload-time = "2026-07-23T01:55:07.383Z" },
+ { url = "https://files.pythonhosted.org/packages/1a/81/d217043a4c17fbce360905e3b2bdd20139ebc9a2de836d035d179c4da006/aiohttp-3.14.3-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:c39846c3aad97a8530c89d7a3869a8f8e9e3762c6ac0504481e5c80948f7e807", size = 1735092, upload-time = "2026-07-23T01:55:09.803Z" },
+ { url = "https://files.pythonhosted.org/packages/a1/66/e13a02d0eeb1a9a502402a977abb4e4abff9fe4051c26f80558c57a7c975/aiohttp-3.14.3-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5895ef58c4620afe02fa16044f023dc4dafec08158f9d08874a46a7dbc0341b8", size = 1800546, upload-time = "2026-07-23T01:55:12.012Z" },
+ { url = "https://files.pythonhosted.org/packages/26/5e/57d42fca1d18cb5acc1cad945d017fabc5d6ae71d8a08ad66be8dc3ee544/aiohttp-3.14.3-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:fa9467a8113aa69d3d7c55a70ef0b7c636010a40993f3df9d9d0d73b3eb7ef24", size = 1895250, upload-time = "2026-07-23T01:55:14.357Z" },
+ { url = "https://files.pythonhosted.org/packages/ca/1c/7da8d08e74d56f00070822f9638ff3f1c563f8ad87d1efa996c87bfc8644/aiohttp-3.14.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d7d2deec16eeedf55f2c7cf75b521ea3856a5177e123844f8fd0f114ce252cb5", size = 1789289, upload-time = "2026-07-23T01:55:16.668Z" },
+ { url = "https://files.pythonhosted.org/packages/cd/0f/cf16bcf56896981c1a0319f5d5db9337994b5165730c48a8fa07e9b34be6/aiohttp-3.14.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:dd54d0e8717de95939766febac482ac0474d8ac3b048115f9f2b1d23a16e7db4", size = 1586706, upload-time = "2026-07-23T01:55:18.913Z" },
+ { url = "https://files.pythonhosted.org/packages/fe/6f/76eac12a7f2480e1e304f842efdb07db33256b0d9165b866b6ef0806c202/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:df82f3787c940c94986b34222d59c9e38843fba85139f36e85255a82ad5355a9", size = 1724652, upload-time = "2026-07-23T01:55:21.296Z" },
+ { url = "https://files.pythonhosted.org/packages/39/b6/19c8c592baeeb94b75f966547d40c02ac7590902306ec5863d5c027cf506/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:42a67efc36300d052fb4508a53e8b6901b9284b599ae63945c377569c5fcc1e1", size = 1756239, upload-time = "2026-07-23T01:55:23.705Z" },
+ { url = "https://files.pythonhosted.org/packages/dc/c9/4e9383150296f97f873b680c4de8fb2cd88608fb9f48c79edcb111611abc/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:7a75aa63cbf9b21cfaf60dc2657e19df2c2867d91707d653fee171ffeedd1371", size = 1769161, upload-time = "2026-07-23T01:55:26.082Z" },
+ { url = "https://files.pythonhosted.org/packages/aa/1e/147bdc6cc5de5f3ab011be8bf5d6e786633249f22c20bae06f85e45f5387/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:e92eb8acc45eb6a9f4935071a77edf5b85cc6f8dfad5cd99e97653c26593cdde", size = 1578759, upload-time = "2026-07-23T01:55:28.846Z" },
+ { url = "https://files.pythonhosted.org/packages/fd/31/78388a9d6040ece2e11df62ea229a822cf5e52d238374b220ae9975b2623/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:b014a6ed7cf912e787149fdc529166d3ceabac23f26efeea3158c9aba2354e7e", size = 1792025, upload-time = "2026-07-23T01:55:31.457Z" },
+ { url = "https://files.pythonhosted.org/packages/03/51/a3d29fdf2c25d796746af8ad6fe56a45d6256c38b0a8a2ed752e1160b3a2/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:3d4f72af88ac2474bb5bca640030320e3d38a0163a1d7533500e87be458eef71", size = 1768477, upload-time = "2026-07-23T01:55:33.87Z" },
+ { url = "https://files.pythonhosted.org/packages/29/a6/442e18b5afeade534d877a2dc3c3e392aff8d49787890b0cf84790410267/aiohttp-3.14.3-cp313-cp313-win32.whl", hash = "sha256:5f08ec777f35ee70720233b8b9811d3bb5d728137f30ac91b7457709c3261ac0", size = 451069, upload-time = "2026-07-23T01:55:36.121Z" },
+ { url = "https://files.pythonhosted.org/packages/9d/69/3d876ac02659f271cf7f6769f14a8e3de5b6e888ed8b5a7e998086a4cec8/aiohttp-3.14.3-cp313-cp313-win_amd64.whl", hash = "sha256:dff9461ec275f22135650d5ba4b4931a11f3958df7dfbb8db630000d4dee0883", size = 476518, upload-time = "2026-07-23T01:55:38.303Z" },
+ { url = "https://files.pythonhosted.org/packages/b2/0e/50d6e6471cd31edce8b282bdec59375a3a69124d8a989a0b1313355cae52/aiohttp-3.14.3-cp313-cp313-win_arm64.whl", hash = "sha256:ddcac3c6b382e81f1dd0499199d4136b877beb4cb5ef770bbbfba56c4b8f55d2", size = 447676, upload-time = "2026-07-23T01:55:40.451Z" },
+ { url = "https://files.pythonhosted.org/packages/c8/20/887fdcf832326571b370ffc347b3e70abe101096f3720126aac161b1d872/aiohttp-3.14.3-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:49f7325beb0f85ef4aef5f48f490269575f83e6e2acad00a1d80b807eb027062", size = 509067, upload-time = "2026-07-23T01:55:42.618Z" },
+ { url = "https://files.pythonhosted.org/packages/ad/a3/92cec936f78cc4bf0fa5554ebe593b73459d94e3c62303e1902a4cccb6f7/aiohttp-3.14.3-cp314-cp314-android_24_x86_64.whl", hash = "sha256:e3be98a7c30b8c25d573dafba7171d66dfb05ee6a9070fc46535464ff97700a6", size = 514774, upload-time = "2026-07-23T01:55:44.937Z" },
+ { url = "https://files.pythonhosted.org/packages/29/ba/2a0c38df3fc557620b6a5acd98364af050053b6285b4dc7ee74100c63c18/aiohttp-3.14.3-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:614c61d478b83953e261d02bb2df750f17227cd33ef8002945bf5aebbde21919", size = 488134, upload-time = "2026-07-23T01:55:47.135Z" },
+ { url = "https://files.pythonhosted.org/packages/48/d6/d51b7d4bf309af3693940d8ffd2b9ed0b682434ef85959b7c9c137f60cf8/aiohttp-3.14.3-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:1caa7b0d05f3e3a36f87788c59e970a7ee1cefcfcbb924a9f138c4a6551c9cb7", size = 494201, upload-time = "2026-07-23T01:55:49.451Z" },
+ { url = "https://files.pythonhosted.org/packages/3f/5a/8f624384e5f1efabb5229b94157eb966b021e97bdb188c62860c2ae243c2/aiohttp-3.14.3-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:dfa68deb2a443bdaa3ea5297b0699c1464f08aef3812b486d1348eee61b07dc0", size = 502766, upload-time = "2026-07-23T01:55:51.656Z" },
+ { url = "https://files.pythonhosted.org/packages/a6/26/4ff0164370deec18fb19254ee4ab10b7a73304ac0c860b13f5f84663759b/aiohttp-3.14.3-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:e72ee89e28d907a18f46959b4eb0bb06701cc7f8cf4366e00029e2ccfaaf5924", size = 756557, upload-time = "2026-07-23T01:55:53.964Z" },
+ { url = "https://files.pythonhosted.org/packages/97/a3/7056b86dc0d9ec709ea9777eae3b0161428f943372f8b98c01c11593b682/aiohttp-3.14.3-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ad4c8b7488d745d2ca4838ebd8ae5ba9b56341d30b1da43640e4ce87f9f49646", size = 510168, upload-time = "2026-07-23T01:55:56.22Z" },
+ { url = "https://files.pythonhosted.org/packages/85/ed/0357a015892fd68058bf2d39d3fd1958e459b997a7db30aaa6aaa434ae96/aiohttp-3.14.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:db332af25642007330fca8be5c4d194caf2bea7a7fc84415aff3497af5dfee6b", size = 512957, upload-time = "2026-07-23T01:55:58.437Z" },
+ { url = "https://files.pythonhosted.org/packages/47/d1/8aba53f15ccb2238405f5e9d30e2a8ca44f93878c26e7165ade00d374b1c/aiohttp-3.14.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:25bd2708db6bdf6a6630dd37bdcdfcb47c4434d22ac69c64665b802910140b30", size = 1750149, upload-time = "2026-07-23T01:56:00.856Z" },
+ { url = "https://files.pythonhosted.org/packages/49/bd/40c3fee327529284375c6701cbb0fa4600cc2e8432af1378f897e2ef7d3a/aiohttp-3.14.3-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:cef89a58e628c4efcac3275c2d68083f82426dcdc89c1492a6f654f9f7ea6ab9", size = 1707685, upload-time = "2026-07-23T01:56:03.371Z" },
+ { url = "https://files.pythonhosted.org/packages/2a/a3/ca0cc6724cca8114b05694abd916060758c79894c3aa5b012cdadc1bc28e/aiohttp-3.14.3-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c23ec8ee9d5ab2f5421f9c7fffce208435607af27fd46d4a44e031954352838f", size = 1803911, upload-time = "2026-07-23T01:56:05.817Z" },
+ { url = "https://files.pythonhosted.org/packages/95/b5/85b099c299c3ffd38ad9b3e43694c8a346934e4a30c88c4fd5a841234f77/aiohttp-3.14.3-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:e2667f0bbe7eb6c74eae5e9691441ad186e5845ca3cff63230fc09c4e7514f5d", size = 1876929, upload-time = "2026-07-23T01:56:08.413Z" },
+ { url = "https://files.pythonhosted.org/packages/d5/b7/1da684a04175473fa4cddbf9a2f572e79514c3fd27a74597f43057d4f3da/aiohttp-3.14.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:18cb43369747b2ae007bd2655fb8e63a099c2ff1d207962943636dac989b3147", size = 1761112, upload-time = "2026-07-23T01:56:10.918Z" },
+ { url = "https://files.pythonhosted.org/packages/d1/16/bc4b55e3e5cb175fd69c53c90d60d2f47797cb343da5106e23863dc4dba4/aiohttp-3.14.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d77640cc618c1d99fc4f8589c0f24a730adfa54eb1e57ef7bf0c8dfb78da898c", size = 1583500, upload-time = "2026-07-23T01:56:13.613Z" },
+ { url = "https://files.pythonhosted.org/packages/2a/e8/13a9d957a1ee40837f46aa30f0f4c657e673ad86a2e6362a9f9be20d26d9/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:53e5179d8abb5710f8e83ba207c41c8d1261fcffd4616500e15ca2b7a33be10a", size = 1713940, upload-time = "2026-07-23T01:56:15.969Z" },
+ { url = "https://files.pythonhosted.org/packages/38/05/d33c680c1bcf1c7e130f9cbfc1fc02fe8bb0c4af2a94a53dd5fb56131e5c/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:cd817772b2fcf2b8c0905795318485f9ec16eae60b29feb7f4c77085311637f0", size = 1724413, upload-time = "2026-07-23T01:56:18.591Z" },
+ { url = "https://files.pythonhosted.org/packages/85/1d/af798d306f7a74b6a632dbcabcf62a4c91391b7582d2a8c6d7712e2cc54e/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:4e3ac92d90e92773b2362d506068e9a948192bd553e743c5b2429e28527c8661", size = 1770748, upload-time = "2026-07-23T01:56:21.074Z" },
+ { url = "https://files.pythonhosted.org/packages/a8/92/ad720d472556a995049206867765e9410969684f86ee09423ff9969044c1/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:3f42e9b78301f11c8f861746175d8b9c1ccef713fcad9eab396e2f6db8ed4a22", size = 1577564, upload-time = "2026-07-23T01:56:23.475Z" },
+ { url = "https://files.pythonhosted.org/packages/60/ad/0ed7586cbef7a884e23a752fa2bb987a122e6a5dd50dab109258d0a95193/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:9d9edccfe496b476db5f398d97b865e9a6752bcf8aec4eef8390ce20fb64bb41", size = 1782080, upload-time = "2026-07-23T01:56:25.994Z" },
+ { url = "https://files.pythonhosted.org/packages/97/ea/dbaed0d73e8a69aad653b045dab451c67c2454bb731a37b45a86593e9422/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:1c5ec8fb1bcc31a8466f74aaf26c345d5c386fa4bd08a3f0eb9c7a4a3fe8b5bf", size = 1745813, upload-time = "2026-07-23T01:56:28.604Z" },
+ { url = "https://files.pythonhosted.org/packages/81/1b/6893d4bc57e434fc93a6c9217c637d967a0b651d989f6e3265179375754a/aiohttp-3.14.3-cp314-cp314-win32.whl", hash = "sha256:38901a84da3ce22249f6e860bf8f90d141bcab7da090cc398f8bb58c0e44b7da", size = 455872, upload-time = "2026-07-23T01:56:31.031Z" },
+ { url = "https://files.pythonhosted.org/packages/f5/8b/c7baa1ba1eda4db6989baefe5de6d99834921b84ebd7918624febcb9f290/aiohttp-3.14.3-cp314-cp314-win_amd64.whl", hash = "sha256:8b3b60de05f3dcb6f6a00f818bb2ec781cee4de0645f59ccaf99b1d1823b6100", size = 481030, upload-time = "2026-07-23T01:56:33.365Z" },
+ { url = "https://files.pythonhosted.org/packages/22/8c/c29d067df825a2df88ca432db848aa2fe8199598359cc06c12b09320cac9/aiohttp-3.14.3-cp314-cp314-win_arm64.whl", hash = "sha256:1576145bdceeb92382d899751e12743a3a5b8e460a841e3e50543859e54864dc", size = 453669, upload-time = "2026-07-23T01:56:35.731Z" },
+ { url = "https://files.pythonhosted.org/packages/6a/a4/9c033beb355d39b6147980597ec9645e4729243f686ee4dc73945de72030/aiohttp-3.14.3-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:8800c996b01c2772a783e3e46f3e1abd5823029adca0df54231960de9bfefa5b", size = 791403, upload-time = "2026-07-23T01:56:37.972Z" },
+ { url = "https://files.pythonhosted.org/packages/80/ca/87c32a0a7704583cfc49660bd817889bae5b830bf53b5dcb4e92145ac2da/aiohttp-3.14.3-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:ebe8e504f058fe91223351cecd2d9d6946c9d241bb0250d898ffbdf584cc72b0", size = 526413, upload-time = "2026-07-23T01:56:40.523Z" },
+ { url = "https://files.pythonhosted.org/packages/9e/d8/8ec0e471248c500acdce2be3f46db8fb62b5eb60efef072529cc85ee1d26/aiohttp-3.14.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:30402d03a7c0ff52bce290b57e564e9079fd9d0cb545c8aba73f86a103162d2e", size = 532135, upload-time = "2026-07-23T01:56:42.876Z" },
+ { url = "https://files.pythonhosted.org/packages/fe/45/f8919fd936e8b79fcd9bda7b6d8e62613462a713f4f17987fd7c34399142/aiohttp-3.14.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9fc7b5bfec6573f3ae844f457fdde5adeb713f8b8e4a81ad64fc207b49383716", size = 1922742, upload-time = "2026-07-23T01:56:45.528Z" },
+ { url = "https://files.pythonhosted.org/packages/f6/ec/9ca76b28a27525b0cc53e20842e0228b022f301ce1f436b7d814b4aaf2df/aiohttp-3.14.3-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:8a5fd34f7f7410d1730d5c2ba873cacb2eed3fede366feb268a70ba22581ed8f", size = 1787371, upload-time = "2026-07-23T01:56:48.045Z" },
+ { url = "https://files.pythonhosted.org/packages/b1/04/6acdbf17315f7b55f1937e3387acb89a3cddeb4995689553d064af8e92ab/aiohttp-3.14.3-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:270d3dace9ca2f10f0da5d8ebe519b7a310fc6112ed916e32df5866df0888553", size = 1912623, upload-time = "2026-07-23T01:56:50.605Z" },
+ { url = "https://files.pythonhosted.org/packages/86/e6/438b0c79ca6f45eb9fd9817dd4c01a91919a38c0de5ee9e05e2b4dc0ece7/aiohttp-3.14.3-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:3ae5b3a59436d089b5395d910121a390feed4d00578eb95a0fd1a329fe963100", size = 2005515, upload-time = "2026-07-23T01:56:53.153Z" },
+ { url = "https://files.pythonhosted.org/packages/bb/6b/62cbd6577758699525f5c712d1ddef57d9875fbab0ae8d5f5a202fd598f8/aiohttp-3.14.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2498f0fe69ead802f9675beca44a7c21c62fdaa4ec5145ea1c3ad6edbee29f85", size = 1879906, upload-time = "2026-07-23T01:56:55.818Z" },
+ { url = "https://files.pythonhosted.org/packages/00/95/18bcbf830a21dc3aae24d8f6b6feaf3db1d2090242d00a7868db2ffb0b67/aiohttp-3.14.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:a0dc483c00da8b673abbb367eb6f8d8f4bcec30eb58529ea13cb42e7fd2dfa33", size = 1675849, upload-time = "2026-07-23T01:56:58.861Z" },
+ { url = "https://files.pythonhosted.org/packages/a9/19/47f4968659c5e23606c3790c80fc624e691c153d036148449ee84d31b287/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:c7d3a97c678d34fc5b59da671ee9cd630096ddc643e7b5a30d54a2a6f3574d3f", size = 1843496, upload-time = "2026-07-23T01:57:01.591Z" },
+ { url = "https://files.pythonhosted.org/packages/64/af/38c33c4dd82fddcb4e56c4653b6f1072a8edbc6b7fa15809f14932c41e2d/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:f8fb78a83c9e5f741ca3a68cfb455c1f5bb83b4e7249a3848b3cd78d0a8563b0", size = 1827746, upload-time = "2026-07-23T01:57:05.131Z" },
+ { url = "https://files.pythonhosted.org/packages/a1/9d/0537cda4885ac8f5b7053d164dd06312f4c483a4edcb8ee5b8aaf2a989bf/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:74ab5b6a9fb13e873e5a90946588baecaf488745e1db1a4a5c433f971f035098", size = 1853810, upload-time = "2026-07-23T01:57:08.043Z" },
+ { url = "https://files.pythonhosted.org/packages/19/fe/26f9c5e6458385aa86497836b0dea6fb2f027827d63f37c7856cce9286ee/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:bd52f811e65f6fb634b1047159657c98f52b407f8efec907bcfc09da9a4c0a25", size = 1668895, upload-time = "2026-07-23T01:57:10.837Z" },
+ { url = "https://files.pythonhosted.org/packages/ec/4c/618b1db9b9ba079b8875d2cdf78e7c4a3bf72903bd5850fee7dd9544600a/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:f0f177d1b195b9e06376cfd7d308d8a1b920909a609d03ac82a8c73bbb16d3b9", size = 1883833, upload-time = "2026-07-23T01:57:13.672Z" },
+ { url = "https://files.pythonhosted.org/packages/94/c6/bd959bd1e4771f9fd944e9e436224c48c77b018b73b519b5aad346335bcc/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:498c6c623134f8e09a3c4e60bcd607a0b4590dd7dbf08dd40851b27cbb520ccb", size = 1844251, upload-time = "2026-07-23T01:57:16.593Z" },
+ { url = "https://files.pythonhosted.org/packages/5e/19/08d41839658bdd44a0ed2480f3891705ecb487ce28c0dde62c9040c997e0/aiohttp-3.14.3-cp314-cp314t-win32.whl", hash = "sha256:b304db572b4368edd8dda8a2274f73156fe15558fca4a917cb8a09fc47af5963", size = 474180, upload-time = "2026-07-23T01:57:19.306Z" },
+ { url = "https://files.pythonhosted.org/packages/99/5d/3cd6ef0a2b2851f7ab913b5b079334781bd50ff56a323e4454063377a080/aiohttp-3.14.3-cp314-cp314t-win_amd64.whl", hash = "sha256:b20032766aedf6261c7a566585a40867d092ac03a0d81592d5370ef9b054f99b", size = 500528, upload-time = "2026-07-23T01:57:21.762Z" },
+ { url = "https://files.pythonhosted.org/packages/a4/37/cfd1ed540a4d318da025590d96b728e63713c09e9377950fc655dadeb856/aiohttp-3.14.3-cp314-cp314t-win_arm64.whl", hash = "sha256:2e1161602f45a54de2ce0905243a95f58cb42dcd378402f3697f5e0b21e9d2e7", size = 469280, upload-time = "2026-07-23T01:57:24.241Z" },
]
[[package]]
@@ -475,15 +475,15 @@ wheels = [
[[package]]
name = "h2"
-version = "4.3.0"
+version = "4.4.1"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "hpack" },
{ name = "hyperframe" },
]
-sdist = { url = "https://files.pythonhosted.org/packages/1d/17/afa56379f94ad0fe8defd37d6eb3f89a25404ffc71d4d848893d270325fc/h2-4.3.0.tar.gz", hash = "sha256:6c59efe4323fa18b47a632221a1888bd7fde6249819beda254aeca909f221bf1", size = 2152026, upload-time = "2025-08-23T18:12:19.778Z" }
+sdist = { url = "https://files.pythonhosted.org/packages/e7/85/7c366e69d84c17bb778fe41419e1fbcce3033d5b7ce29bbffff0a98b859f/h2-4.4.1.tar.gz", hash = "sha256:4e866ffb1a869ae14dd9b5e6beb5c24a13da0495ad72b65925ded182521c1516", size = 2157281, upload-time = "2026-08-03T11:45:09.509Z" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/69/b2/119f6e6dcbd96f9069ce9a2665e0146588dc9f88f29549711853645e736a/h2-4.3.0-py3-none-any.whl", hash = "sha256:c438f029a25f7945c69e0ccf0fb951dc3f73a5f6412981daee861431b70e2bdd", size = 61779, upload-time = "2025-08-23T18:12:17.779Z" },
+ { url = "https://files.pythonhosted.org/packages/7e/22/e85faf23bd72a92d1921e37d674ca56eb298a3c8be31fdecef0ff2b3aaac/h2-4.4.1-py3-none-any.whl", hash = "sha256:0e25f1462b23c9cb82d9eb02e28bc706dac2a68cb457c6a0d74d63c8a2a5d0e6", size = 62636, upload-time = "2026-08-03T11:44:59.164Z" },
]
[[package]]
@@ -517,11 +517,11 @@ wheels = [
[[package]]
name = "hpack"
-version = "4.1.0"
+version = "4.2.0"
source = { registry = "https://pypi.org/simple" }
-sdist = { url = "https://files.pythonhosted.org/packages/2c/48/71de9ed269fdae9c8057e5a4c0aa7402e8bb16f2c6e90b3aa53327b113f8/hpack-4.1.0.tar.gz", hash = "sha256:ec5eca154f7056aa06f196a557655c5b009b382873ac8d1e66e79e87535f1dca", size = 51276, upload-time = "2025-01-22T21:44:58.347Z" }
+sdist = { url = "https://files.pythonhosted.org/packages/26/5b/fcabf6028144a8723726318b07a32c2f3314acdff6265743cf08a344b18e/hpack-4.2.0.tar.gz", hash = "sha256:0895cfa3b5531fc65fe439c05eb65144f123bf7a394fcaa56aa423548d8e45c0", size = 51300, upload-time = "2026-06-23T18:34:46.667Z" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/07/c6/80c95b1b2b94682a72cbdbfb85b81ae2daffa4291fbfa1b1464502ede10d/hpack-4.1.0-py3-none-any.whl", hash = "sha256:157ac792668d995c657d93111f46b4535ed114f0c9c8d672271bbec7eae1b496", size = 34357, upload-time = "2025-01-22T21:44:56.92Z" },
+ { url = "https://files.pythonhosted.org/packages/71/b4/4a9fcfb2aef6ba44d9073ecd301443aa00b3dac95de5619f2a7de7ec8a91/hpack-4.2.0-py3-none-any.whl", hash = "sha256:858ac0b02280fa582b5080d68db0899c62a80375e0e5413a74970c5e518b6986", size = 34246, upload-time = "2026-06-23T18:34:45.472Z" },
]
[[package]]
@@ -1018,71 +1018,73 @@ wheels = [
[[package]]
name = "pillow"
-version = "12.2.0"
+version = "12.3.0"
source = { registry = "https://pypi.org/simple" }
-sdist = { url = "https://files.pythonhosted.org/packages/8c/21/c2bcdd5906101a30244eaffc1b6e6ce71a31bd0742a01eb89e660ebfac2d/pillow-12.2.0.tar.gz", hash = "sha256:a830b1a40919539d07806aa58e1b114df53ddd43213d9c8b75847eee6c0182b5", size = 46987819, upload-time = "2026-04-01T14:46:17.687Z" }
+sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/58/be/7482c8a5ebebbc6470b3eb791812fff7d5e0216c2be3827b30b8bb6603ed/pillow-12.2.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:2d192a155bbcec180f8564f693e6fd9bccff5a7af9b32e2e4bf8c9c69dbad6b5", size = 5308279, upload-time = "2026-04-01T14:43:13.246Z" },
- { url = "https://files.pythonhosted.org/packages/d8/95/0a351b9289c2b5cbde0bacd4a83ebc44023e835490a727b2a3bd60ddc0f4/pillow-12.2.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f3f40b3c5a968281fd507d519e444c35f0ff171237f4fdde090dd60699458421", size = 4695490, upload-time = "2026-04-01T14:43:15.584Z" },
- { url = "https://files.pythonhosted.org/packages/de/af/4e8e6869cbed569d43c416fad3dc4ecb944cb5d9492defaed89ddd6fe871/pillow-12.2.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:03e7e372d5240cc23e9f07deca4d775c0817bffc641b01e9c3af208dbd300987", size = 6284462, upload-time = "2026-04-01T14:43:18.268Z" },
- { url = "https://files.pythonhosted.org/packages/e9/9e/c05e19657fd57841e476be1ab46c4d501bffbadbafdc31a6d665f8b737b6/pillow-12.2.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:b86024e52a1b269467a802258c25521e6d742349d760728092e1bc2d135b4d76", size = 8094744, upload-time = "2026-04-01T14:43:20.716Z" },
- { url = "https://files.pythonhosted.org/packages/2b/54/1789c455ed10176066b6e7e6da1b01e50e36f94ba584dc68d9eebfe9156d/pillow-12.2.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7371b48c4fa448d20d2714c9a1f775a81155050d383333e0a6c15b1123dda005", size = 6398371, upload-time = "2026-04-01T14:43:23.443Z" },
- { url = "https://files.pythonhosted.org/packages/43/e3/fdc657359e919462369869f1c9f0e973f353f9a9ee295a39b1fea8ee1a77/pillow-12.2.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:62f5409336adb0663b7caa0da5c7d9e7bdbaae9ce761d34669420c2a801b2780", size = 7087215, upload-time = "2026-04-01T14:43:26.758Z" },
- { url = "https://files.pythonhosted.org/packages/8b/f8/2f6825e441d5b1959d2ca5adec984210f1ec086435b0ed5f52c19b3b8a6e/pillow-12.2.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:01afa7cf67f74f09523699b4e88c73fb55c13346d212a59a2db1f86b0a63e8c5", size = 6509783, upload-time = "2026-04-01T14:43:29.56Z" },
- { url = "https://files.pythonhosted.org/packages/67/f9/029a27095ad20f854f9dba026b3ea6428548316e057e6fc3545409e86651/pillow-12.2.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:fc3d34d4a8fbec3e88a79b92e5465e0f9b842b628675850d860b8bd300b159f5", size = 7212112, upload-time = "2026-04-01T14:43:32.091Z" },
- { url = "https://files.pythonhosted.org/packages/be/42/025cfe05d1be22dbfdb4f264fe9de1ccda83f66e4fc3aac94748e784af04/pillow-12.2.0-cp312-cp312-win32.whl", hash = "sha256:58f62cc0f00fd29e64b29f4fd923ffdb3859c9f9e6105bfc37ba1d08994e8940", size = 6378489, upload-time = "2026-04-01T14:43:34.601Z" },
- { url = "https://files.pythonhosted.org/packages/5d/7b/25a221d2c761c6a8ae21bfa3874988ff2583e19cf8a27bf2fee358df7942/pillow-12.2.0-cp312-cp312-win_amd64.whl", hash = "sha256:7f84204dee22a783350679a0333981df803dac21a0190d706a50475e361c93f5", size = 7084129, upload-time = "2026-04-01T14:43:37.213Z" },
- { url = "https://files.pythonhosted.org/packages/10/e1/542a474affab20fd4a0f1836cb234e8493519da6b76899e30bcc5d990b8b/pillow-12.2.0-cp312-cp312-win_arm64.whl", hash = "sha256:af73337013e0b3b46f175e79492d96845b16126ddf79c438d7ea7ff27783a414", size = 2463612, upload-time = "2026-04-01T14:43:39.421Z" },
- { url = "https://files.pythonhosted.org/packages/4a/01/53d10cf0dbad820a8db274d259a37ba50b88b24768ddccec07355382d5ad/pillow-12.2.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:8297651f5b5679c19968abefd6bb84d95fe30ef712eb1b2d9b2d31ca61267f4c", size = 4100837, upload-time = "2026-04-01T14:43:41.506Z" },
- { url = "https://files.pythonhosted.org/packages/0f/98/f3a6657ecb698c937f6c76ee564882945f29b79bad496abcba0e84659ec5/pillow-12.2.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:50d8520da2a6ce0af445fa6d648c4273c3eeefbc32d7ce049f22e8b5c3daecc2", size = 4176528, upload-time = "2026-04-01T14:43:43.773Z" },
- { url = "https://files.pythonhosted.org/packages/69/bc/8986948f05e3ea490b8442ea1c1d4d990b24a7e43d8a51b2c7d8b1dced36/pillow-12.2.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:766cef22385fa1091258ad7e6216792b156dc16d8d3fa607e7545b2b72061f1c", size = 3640401, upload-time = "2026-04-01T14:43:45.87Z" },
- { url = "https://files.pythonhosted.org/packages/34/46/6c717baadcd62bc8ed51d238d521ab651eaa74838291bda1f86fe1f864c9/pillow-12.2.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:5d2fd0fa6b5d9d1de415060363433f28da8b1526c1c129020435e186794b3795", size = 5308094, upload-time = "2026-04-01T14:43:48.438Z" },
- { url = "https://files.pythonhosted.org/packages/71/43/905a14a8b17fdb1ccb58d282454490662d2cb89a6bfec26af6d3520da5ec/pillow-12.2.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:56b25336f502b6ed02e889f4ece894a72612fe885889a6e8c4c80239ff6e5f5f", size = 4695402, upload-time = "2026-04-01T14:43:51.292Z" },
- { url = "https://files.pythonhosted.org/packages/73/dd/42107efcb777b16fa0393317eac58f5b5cf30e8392e266e76e51cff28c3d/pillow-12.2.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:f1c943e96e85df3d3478f7b691f229887e143f81fedab9b20205349ab04d73ed", size = 6280005, upload-time = "2026-04-01T14:43:54.242Z" },
- { url = "https://files.pythonhosted.org/packages/a8/68/b93e09e5e8549019e61acf49f65b1a8530765a7f812c77a7461bca7e4494/pillow-12.2.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:03f6fab9219220f041c74aeaa2939ff0062bd5c364ba9ce037197f4c6d498cd9", size = 8090669, upload-time = "2026-04-01T14:43:57.335Z" },
- { url = "https://files.pythonhosted.org/packages/4b/6e/3ccb54ce8ec4ddd1accd2d89004308b7b0b21c4ac3d20fa70af4760a4330/pillow-12.2.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5cdfebd752ec52bf5bb4e35d9c64b40826bc5b40a13df7c3cda20a2c03a0f5ed", size = 6395194, upload-time = "2026-04-01T14:43:59.864Z" },
- { url = "https://files.pythonhosted.org/packages/67/ee/21d4e8536afd1a328f01b359b4d3997b291ffd35a237c877b331c1c3b71c/pillow-12.2.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:eedf4b74eda2b5a4b2b2fb4c006d6295df3bf29e459e198c90ea48e130dc75c3", size = 7082423, upload-time = "2026-04-01T14:44:02.74Z" },
- { url = "https://files.pythonhosted.org/packages/78/5f/e9f86ab0146464e8c133fe85df987ed9e77e08b29d8d35f9f9f4d6f917ba/pillow-12.2.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:00a2865911330191c0b818c59103b58a5e697cae67042366970a6b6f1b20b7f9", size = 6505667, upload-time = "2026-04-01T14:44:05.381Z" },
- { url = "https://files.pythonhosted.org/packages/ed/1e/409007f56a2fdce61584fd3acbc2bbc259857d555196cedcadc68c015c82/pillow-12.2.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:1e1757442ed87f4912397c6d35a0db6a7b52592156014706f17658ff58bbf795", size = 7208580, upload-time = "2026-04-01T14:44:08.39Z" },
- { url = "https://files.pythonhosted.org/packages/23/c4/7349421080b12fb35414607b8871e9534546c128a11965fd4a7002ccfbee/pillow-12.2.0-cp313-cp313-win32.whl", hash = "sha256:144748b3af2d1b358d41286056d0003f47cb339b8c43a9ea42f5fea4d8c66b6e", size = 6375896, upload-time = "2026-04-01T14:44:11.197Z" },
- { url = "https://files.pythonhosted.org/packages/3f/82/8a3739a5e470b3c6cbb1d21d315800d8e16bff503d1f16b03a4ec3212786/pillow-12.2.0-cp313-cp313-win_amd64.whl", hash = "sha256:390ede346628ccc626e5730107cde16c42d3836b89662a115a921f28440e6a3b", size = 7081266, upload-time = "2026-04-01T14:44:13.947Z" },
- { url = "https://files.pythonhosted.org/packages/c3/25/f968f618a062574294592f668218f8af564830ccebdd1fa6200f598e65c5/pillow-12.2.0-cp313-cp313-win_arm64.whl", hash = "sha256:8023abc91fba39036dbce14a7d6535632f99c0b857807cbbbf21ecc9f4717f06", size = 2463508, upload-time = "2026-04-01T14:44:16.312Z" },
- { url = "https://files.pythonhosted.org/packages/4d/a4/b342930964e3cb4dce5038ae34b0eab4653334995336cd486c5a8c25a00c/pillow-12.2.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:042db20a421b9bafecc4b84a8b6e444686bd9d836c7fd24542db3e7df7baad9b", size = 5309927, upload-time = "2026-04-01T14:44:18.89Z" },
- { url = "https://files.pythonhosted.org/packages/9f/de/23198e0a65a9cf06123f5435a5d95cea62a635697f8f03d134d3f3a96151/pillow-12.2.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:dd025009355c926a84a612fecf58bb315a3f6814b17ead51a8e48d3823d9087f", size = 4698624, upload-time = "2026-04-01T14:44:21.115Z" },
- { url = "https://files.pythonhosted.org/packages/01/a6/1265e977f17d93ea37aa28aa81bad4fa597933879fac2520d24e021c8da3/pillow-12.2.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:88ddbc66737e277852913bd1e07c150cc7bb124539f94c4e2df5344494e0a612", size = 6321252, upload-time = "2026-04-01T14:44:23.663Z" },
- { url = "https://files.pythonhosted.org/packages/3c/83/5982eb4a285967baa70340320be9f88e57665a387e3a53a7f0db8231a0cd/pillow-12.2.0-cp313-cp313t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:d362d1878f00c142b7e1a16e6e5e780f02be8195123f164edf7eddd911eefe7c", size = 8126550, upload-time = "2026-04-01T14:44:26.772Z" },
- { url = "https://files.pythonhosted.org/packages/4e/48/6ffc514adce69f6050d0753b1a18fd920fce8cac87620d5a31231b04bfc5/pillow-12.2.0-cp313-cp313t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2c727a6d53cb0018aadd8018c2b938376af27914a68a492f59dfcaca650d5eea", size = 6433114, upload-time = "2026-04-01T14:44:29.615Z" },
- { url = "https://files.pythonhosted.org/packages/36/a3/f9a77144231fb8d40ee27107b4463e205fa4677e2ca2548e14da5cf18dce/pillow-12.2.0-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:efd8c21c98c5cc60653bcb311bef2ce0401642b7ce9d09e03a7da87c878289d4", size = 7115667, upload-time = "2026-04-01T14:44:32.773Z" },
- { url = "https://files.pythonhosted.org/packages/c1/fc/ac4ee3041e7d5a565e1c4fd72a113f03b6394cc72ab7089d27608f8aaccb/pillow-12.2.0-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:9f08483a632889536b8139663db60f6724bfcb443c96f1b18855860d7d5c0fd4", size = 6538966, upload-time = "2026-04-01T14:44:35.252Z" },
- { url = "https://files.pythonhosted.org/packages/c0/a8/27fb307055087f3668f6d0a8ccb636e7431d56ed0750e07a60547b1e083e/pillow-12.2.0-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:dac8d77255a37e81a2efcbd1fc05f1c15ee82200e6c240d7e127e25e365c39ea", size = 7238241, upload-time = "2026-04-01T14:44:37.875Z" },
- { url = "https://files.pythonhosted.org/packages/ad/4b/926ab182c07fccae9fcb120043464e1ff1564775ec8864f21a0ebce6ac25/pillow-12.2.0-cp313-cp313t-win32.whl", hash = "sha256:ee3120ae9dff32f121610bb08e4313be87e03efeadfc6c0d18f89127e24d0c24", size = 6379592, upload-time = "2026-04-01T14:44:40.336Z" },
- { url = "https://files.pythonhosted.org/packages/c2/c4/f9e476451a098181b30050cc4c9a3556b64c02cf6497ea421ac047e89e4b/pillow-12.2.0-cp313-cp313t-win_amd64.whl", hash = "sha256:325ca0528c6788d2a6c3d40e3568639398137346c3d6e66bb61db96b96511c98", size = 7085542, upload-time = "2026-04-01T14:44:43.251Z" },
- { url = "https://files.pythonhosted.org/packages/00/a4/285f12aeacbe2d6dc36c407dfbbe9e96d4a80b0fb710a337f6d2ad978c75/pillow-12.2.0-cp313-cp313t-win_arm64.whl", hash = "sha256:2e5a76d03a6c6dcef67edabda7a52494afa4035021a79c8558e14af25313d453", size = 2465765, upload-time = "2026-04-01T14:44:45.996Z" },
- { url = "https://files.pythonhosted.org/packages/bf/98/4595daa2365416a86cb0d495248a393dfc84e96d62ad080c8546256cb9c0/pillow-12.2.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:3adc9215e8be0448ed6e814966ecf3d9952f0ea40eb14e89a102b87f450660d8", size = 4100848, upload-time = "2026-04-01T14:44:48.48Z" },
- { url = "https://files.pythonhosted.org/packages/0b/79/40184d464cf89f6663e18dfcf7ca21aae2491fff1a16127681bf1fa9b8cf/pillow-12.2.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:6a9adfc6d24b10f89588096364cc726174118c62130c817c2837c60cf08a392b", size = 4176515, upload-time = "2026-04-01T14:44:51.353Z" },
- { url = "https://files.pythonhosted.org/packages/b0/63/703f86fd4c422a9cf722833670f4f71418fb116b2853ff7da722ea43f184/pillow-12.2.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:6a6e67ea2e6feda684ed370f9a1c52e7a243631c025ba42149a2cc5934dec295", size = 3640159, upload-time = "2026-04-01T14:44:53.588Z" },
- { url = "https://files.pythonhosted.org/packages/71/e0/fb22f797187d0be2270f83500aab851536101b254bfa1eae10795709d283/pillow-12.2.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:2bb4a8d594eacdfc59d9e5ad972aa8afdd48d584ffd5f13a937a664c3e7db0ed", size = 5312185, upload-time = "2026-04-01T14:44:56.039Z" },
- { url = "https://files.pythonhosted.org/packages/ba/8c/1a9e46228571de18f8e28f16fabdfc20212a5d019f3e3303452b3f0a580d/pillow-12.2.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:80b2da48193b2f33ed0c32c38140f9d3186583ce7d516526d462645fd98660ae", size = 4695386, upload-time = "2026-04-01T14:44:58.663Z" },
- { url = "https://files.pythonhosted.org/packages/70/62/98f6b7f0c88b9addd0e87c217ded307b36be024d4ff8869a812b241d1345/pillow-12.2.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:22db17c68434de69d8ecfc2fe821569195c0c373b25cccb9cbdacf2c6e53c601", size = 6280384, upload-time = "2026-04-01T14:45:01.5Z" },
- { url = "https://files.pythonhosted.org/packages/5e/03/688747d2e91cfbe0e64f316cd2e8005698f76ada3130d0194664174fa5de/pillow-12.2.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:7b14cc0106cd9aecda615dd6903840a058b4700fcb817687d0ee4fc8b6e389be", size = 8091599, upload-time = "2026-04-01T14:45:04.5Z" },
- { url = "https://files.pythonhosted.org/packages/f6/35/577e22b936fcdd66537329b33af0b4ccfefaeabd8aec04b266528cddb33c/pillow-12.2.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:8cbeb542b2ebc6fcdacabf8aca8c1a97c9b3ad3927d46b8723f9d4f033288a0f", size = 6396021, upload-time = "2026-04-01T14:45:07.117Z" },
- { url = "https://files.pythonhosted.org/packages/11/8d/d2532ad2a603ca2b93ad9f5135732124e57811d0168155852f37fbce2458/pillow-12.2.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4bfd07bc812fbd20395212969e41931001fd59eb55a60658b0e5710872e95286", size = 7083360, upload-time = "2026-04-01T14:45:09.763Z" },
- { url = "https://files.pythonhosted.org/packages/5e/26/d325f9f56c7e039034897e7380e9cc202b1e368bfd04d4cbe6a441f02885/pillow-12.2.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:9aba9a17b623ef750a4d11b742cbafffeb48a869821252b30ee21b5e91392c50", size = 6507628, upload-time = "2026-04-01T14:45:12.378Z" },
- { url = "https://files.pythonhosted.org/packages/5f/f7/769d5632ffb0988f1c5e7660b3e731e30f7f8ec4318e94d0a5d674eb65a4/pillow-12.2.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:deede7c263feb25dba4e82ea23058a235dcc2fe1f6021025dc71f2b618e26104", size = 7209321, upload-time = "2026-04-01T14:45:15.122Z" },
- { url = "https://files.pythonhosted.org/packages/6a/7a/c253e3c645cd47f1aceea6a8bacdba9991bf45bb7dfe927f7c893e89c93c/pillow-12.2.0-cp314-cp314-win32.whl", hash = "sha256:632ff19b2778e43162304d50da0181ce24ac5bb8180122cbe1bf4673428328c7", size = 6479723, upload-time = "2026-04-01T14:45:17.797Z" },
- { url = "https://files.pythonhosted.org/packages/cd/8b/601e6566b957ca50e28725cb6c355c59c2c8609751efbecd980db44e0349/pillow-12.2.0-cp314-cp314-win_amd64.whl", hash = "sha256:4e6c62e9d237e9b65fac06857d511e90d8461a32adcc1b9065ea0c0fa3a28150", size = 7217400, upload-time = "2026-04-01T14:45:20.529Z" },
- { url = "https://files.pythonhosted.org/packages/d6/94/220e46c73065c3e2951bb91c11a1fb636c8c9ad427ac3ce7d7f3359b9b2f/pillow-12.2.0-cp314-cp314-win_arm64.whl", hash = "sha256:b1c1fbd8a5a1af3412a0810d060a78b5136ec0836c8a4ef9aa11807f2a22f4e1", size = 2554835, upload-time = "2026-04-01T14:45:23.162Z" },
- { url = "https://files.pythonhosted.org/packages/b6/ab/1b426a3974cb0e7da5c29ccff4807871d48110933a57207b5a676cccc155/pillow-12.2.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:57850958fe9c751670e49b2cecf6294acc99e562531f4bd317fa5ddee2068463", size = 5314225, upload-time = "2026-04-01T14:45:25.637Z" },
- { url = "https://files.pythonhosted.org/packages/19/1e/dce46f371be2438eecfee2a1960ee2a243bbe5e961890146d2dee1ff0f12/pillow-12.2.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:d5d38f1411c0ed9f97bcb49b7bd59b6b7c314e0e27420e34d99d844b9ce3b6f3", size = 4698541, upload-time = "2026-04-01T14:45:28.355Z" },
- { url = "https://files.pythonhosted.org/packages/55/c3/7fbecf70adb3a0c33b77a300dc52e424dc22ad8cdc06557a2e49523b703d/pillow-12.2.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:5c0a9f29ca8e79f09de89293f82fc9b0270bb4af1d58bc98f540cc4aedf03166", size = 6322251, upload-time = "2026-04-01T14:45:30.924Z" },
- { url = "https://files.pythonhosted.org/packages/1c/3c/7fbc17cfb7e4fe0ef1642e0abc17fc6c94c9f7a16be41498e12e2ba60408/pillow-12.2.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1610dd6c61621ae1cf811bef44d77e149ce3f7b95afe66a4512f8c59f25d9ebe", size = 8127807, upload-time = "2026-04-01T14:45:33.908Z" },
- { url = "https://files.pythonhosted.org/packages/ff/c3/a8ae14d6defd2e448493ff512fae903b1e9bd40b72efb6ec55ce0048c8ce/pillow-12.2.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0a34329707af4f73cf1782a36cd2289c0368880654a2c11f027bcee9052d35dd", size = 6433935, upload-time = "2026-04-01T14:45:36.623Z" },
- { url = "https://files.pythonhosted.org/packages/6e/32/2880fb3a074847ac159d8f902cb43278a61e85f681661e7419e6596803ed/pillow-12.2.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8e9c4f5b3c546fa3458a29ab22646c1c6c787ea8f5ef51300e5a60300736905e", size = 7116720, upload-time = "2026-04-01T14:45:39.258Z" },
- { url = "https://files.pythonhosted.org/packages/46/87/495cc9c30e0129501643f24d320076f4cc54f718341df18cc70ec94c44e1/pillow-12.2.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:fb043ee2f06b41473269765c2feae53fc2e2fbf96e5e22ca94fb5ad677856f06", size = 6540498, upload-time = "2026-04-01T14:45:41.879Z" },
- { url = "https://files.pythonhosted.org/packages/18/53/773f5edca692009d883a72211b60fdaf8871cbef075eaa9d577f0a2f989e/pillow-12.2.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:f278f034eb75b4e8a13a54a876cc4a5ab39173d2cdd93a638e1b467fc545ac43", size = 7239413, upload-time = "2026-04-01T14:45:44.705Z" },
- { url = "https://files.pythonhosted.org/packages/c9/e4/4b64a97d71b2a83158134abbb2f5bd3f8a2ea691361282f010998f339ec7/pillow-12.2.0-cp314-cp314t-win32.whl", hash = "sha256:6bb77b2dcb06b20f9f4b4a8454caa581cd4dd0643a08bacf821216a16d9c8354", size = 6482084, upload-time = "2026-04-01T14:45:47.568Z" },
- { url = "https://files.pythonhosted.org/packages/ba/13/306d275efd3a3453f72114b7431c877d10b1154014c1ebbedd067770d629/pillow-12.2.0-cp314-cp314t-win_amd64.whl", hash = "sha256:6562ace0d3fb5f20ed7290f1f929cae41b25ae29528f2af1722966a0a02e2aa1", size = 7225152, upload-time = "2026-04-01T14:45:50.032Z" },
- { url = "https://files.pythonhosted.org/packages/ff/6e/cf826fae916b8658848d7b9f38d88da6396895c676e8086fc0988073aaf8/pillow-12.2.0-cp314-cp314t-win_arm64.whl", hash = "sha256:aa88ccfe4e32d362816319ed727a004423aab09c5cea43c01a4b435643fa34eb", size = 2556579, upload-time = "2026-04-01T14:45:52.529Z" },
+ { url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" },
+ { url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" },
+ { url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" },
+ { url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" },
+ { url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" },
+ { url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" },
+ { url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" },
+ { url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" },
+ { url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" },
+ { url = "https://files.pythonhosted.org/packages/9d/ac/31fb64e1e7efb5a4b50cd3d92049ba89ac6e4d8d3bb6a74e15048ca3353e/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89", size = 4161684, upload-time = "2026-07-01T11:54:25.934Z" },
+ { url = "https://files.pythonhosted.org/packages/87/b4/9805e23d2b4d77842b468513841fda254ee42f0289d25088340e4ff46e2d/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace", size = 4255487, upload-time = "2026-07-01T11:54:27.935Z" },
+ { url = "https://files.pythonhosted.org/packages/df/39/ecf519435a200c693fe053a6ee4d835b41cf963a4dfc2551c4e637cb2a71/pillow-12.3.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec", size = 3696433, upload-time = "2026-07-01T11:54:29.813Z" },
+ { url = "https://files.pythonhosted.org/packages/42/92/2fc3ffad878ae8dd5469ec1bc8eb83b71f48e13efdf68f02709003982a32/pillow-12.3.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66", size = 5345889, upload-time = "2026-07-01T11:54:31.97Z" },
+ { url = "https://files.pythonhosted.org/packages/10/76/8803c13605b763d33d156c4678fc77f8443389c0c51c8aef707bb02015f4/pillow-12.3.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35", size = 4780109, upload-time = "2026-07-01T11:54:34.026Z" },
+ { url = "https://files.pythonhosted.org/packages/1f/01/e18aff37cb0b4aac47ac90f016d347a49aca667ef97f190b06ac2aabc928/pillow-12.3.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65", size = 6263736, upload-time = "2026-07-01T11:54:36.131Z" },
+ { url = "https://files.pythonhosted.org/packages/f7/62/de5bdd77d935331f4f802edc11e4d82950f642caad6cb2f949837b8560e2/pillow-12.3.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3", size = 6937129, upload-time = "2026-07-01T11:54:38.216Z" },
+ { url = "https://files.pythonhosted.org/packages/70/4d/105627a13300c5e0df1d174230b32fd1273062c96f7745fd552b945d1e1d/pillow-12.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a", size = 6339562, upload-time = "2026-07-01T11:54:40.354Z" },
+ { url = "https://files.pythonhosted.org/packages/6b/1d/f13de01a553988ab895ba1c722e06cf3144d4f57656fd5b81b6d881f1179/pillow-12.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e", size = 7049439, upload-time = "2026-07-01T11:54:42.489Z" },
+ { url = "https://files.pythonhosted.org/packages/c9/f9/066794cca041b969964f779ee5fa66a9498bbf34248ac39c5d7954e4198f/pillow-12.3.0-cp313-cp313-win32.whl", hash = "sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f", size = 6473287, upload-time = "2026-07-01T11:54:44.9Z" },
+ { url = "https://files.pythonhosted.org/packages/a6/9b/7a58e61d62be561da3a356fe2384d4059a6345fc130e23ef1c36a5b81d24/pillow-12.3.0-cp313-cp313-win_amd64.whl", hash = "sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8", size = 7239691, upload-time = "2026-07-01T11:54:47.141Z" },
+ { url = "https://files.pythonhosted.org/packages/aa/b0/c4ed4f0ef8f8fa5ee8351537db6650bb8189f7e118842978dd6589065692/pillow-12.3.0-cp313-cp313-win_arm64.whl", hash = "sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b", size = 2568185, upload-time = "2026-07-01T11:54:49.137Z" },
+ { url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" },
+ { url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" },
+ { url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" },
+ { url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" },
+ { url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" },
+ { url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" },
+ { url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" },
+ { url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" },
+ { url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" },
+ { url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" },
+ { url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" },
+ { url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" },
+ { url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" },
+ { url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" },
+ { url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" },
+ { url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" },
+ { url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" },
+ { url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" },
+ { url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" },
+ { url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" },
+ { url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" },
+ { url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" },
+ { url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" },
+ { url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" },
+ { url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" },
+ { url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" },
+ { url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" },
+ { url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" },
+ { url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" },
+ { url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" },
+ { url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" },
+ { url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" },
+ { url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" },
+ { url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" },
+ { url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" },
+ { url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" },
+ { url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" },
+ { url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" },
+ { url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" },
+ { url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" },
+ { url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" },
+ { url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" },
]
[[package]]
@@ -1438,8 +1440,8 @@ wheels = [
[[package]]
name = "qdrant-client"
-version = "1.18.0"
-source = { git = "https://github.com/qdrant/qdrant-client?tag=v1.18.0#326adefcc2158121dd0d04877e1a483b5aa2627b" }
+version = "1.19.0"
+source = { git = "https://github.com/qdrant/qdrant-client?tag=v1.19.0#425840be987cd470d19bbb4e2363e87754fbc914" }
dependencies = [
{ name = "grpcio" },
{ name = "httpx", extra = ["http2"] },
@@ -1452,17 +1454,17 @@ dependencies = [
[[package]]
name = "qdrant-edge-py"
-version = "0.7.2"
+version = "0.8.0"
source = { registry = "https://pypi.org/simple" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/56/cb/94de5aebf8380172da89c4028613146cf60da89b37526953495a3e938d67/qdrant_edge_py-0.7.2-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:55198076f0de80330737b53cee99edf95faeffb00aa9bca6bc302e1d601d20b6", size = 10526152, upload-time = "2026-06-01T10:20:58.708Z" },
- { url = "https://files.pythonhosted.org/packages/0a/b2/21559919e38a039b0d1be85c024b5781cabddb8d50a28250880bc1cab503/qdrant_edge_py-0.7.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:fc34918fde9e3762c12c228a7b4e3baaaca6b3f3ddb9c9ae8092090f91d8b761", size = 9891156, upload-time = "2026-06-01T10:21:00.464Z" },
- { url = "https://files.pythonhosted.org/packages/d1/32/7fc9515d9645457effd596b5fc018510475a9f7b6d1b02f90593d507a17f/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:1157924bdcb081af04d2f21c7658e4120e7793b214ac81f77750b359cf921689", size = 11261757, upload-time = "2026-06-01T10:21:02.564Z" },
- { url = "https://files.pythonhosted.org/packages/40/ec/2a475eda7f6b83a1a89f2fc41fa2c08294e0f71f735fc45148144a0185c0/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:c8f070ca8c49e2175b405794c490340638c841459cbbacb79e873bb286f9e9ae", size = 10651844, upload-time = "2026-06-01T10:21:04.442Z" },
- { url = "https://files.pythonhosted.org/packages/d2/4d/5d3c8ea787f2ffd47d997c575f196e7b7486db404fa6b6fe522be5a10d13/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:f2ecd4cbe6dedfa408b933bfe28917d9274cdfa58243d32ca3c390cd3b1c721e", size = 11259962, upload-time = "2026-06-01T10:21:06.367Z" },
- { url = "https://files.pythonhosted.org/packages/57/ae/18c6990c3a58628f97381e1494916120350fcf05c4d3dfcf82433cedb9a9/qdrant_edge_py-0.7.2-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:93b8c0b4f7d6fc58284dc0533d66bf18742767c1fbaef60ddd47651a79137e57", size = 10825604, upload-time = "2026-06-01T10:21:08.404Z" },
- { url = "https://files.pythonhosted.org/packages/a1/eb/a79ca27405196ad3da8f1fe88c7dc5b3888e25d531c6aed043c5efa48124/qdrant_edge_py-0.7.2-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b13a892947b5eb2e8ac1a8049fd57a88d1b1d0a390ae2c432d1c0c8b8b0e3c0d", size = 11481427, upload-time = "2026-06-01T10:21:10.604Z" },
- { url = "https://files.pythonhosted.org/packages/dc/45/e7b1f28f82ba2392bc893ef65f20b8093c917396493f1ca43c1b2db56c09/qdrant_edge_py-0.7.2-cp310-abi3-win_amd64.whl", hash = "sha256:e6642e1f73ff28f6aeb2c377fdbe93dd0fa25db2d0c54936279e77e2d8ffb574", size = 10599513, upload-time = "2026-06-01T10:21:12.703Z" },
+ { url = "https://files.pythonhosted.org/packages/95/9a/b8dfc0e6c81797ae437a49ead18d9e95d995861c131dac41bf4a3a3cd93f/qdrant_edge_py-0.8.0-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:8fd85325c350723c6f0fed4a0a6f6dc02488a64cbad4abce554da916a84382dd", size = 11173604, upload-time = "2026-08-05T13:25:12.783Z" },
+ { url = "https://files.pythonhosted.org/packages/4f/d6/9b52178490526866aeff57bec2aa28972a85fc532da20d79db3a82a4eaf3/qdrant_edge_py-0.8.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d84d0702a31b6560c84f4d28b3272f94f1df354c04b80167e73acc08cc134642", size = 10016514, upload-time = "2026-08-05T13:25:14.958Z" },
+ { url = "https://files.pythonhosted.org/packages/c0/14/c42c108fc969aacf3bb084fe20c6cb0fa4b3278fdf731e8ba0325e69d5f5/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:652d754cebd12597a4e1f1fa5305e77e6ba7e056153409a2006f82cbb600b056", size = 12164107, upload-time = "2026-08-05T13:25:16.758Z" },
+ { url = "https://files.pythonhosted.org/packages/de/1e/24c7924f398c05efa87895fa5a6b7027949139d0e17b1089c961d506aa43/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:e6a712253c5056ffb0c8f59bd0cde686bbe9dffe697b6ca881a51c21ff8626d0", size = 11560725, upload-time = "2026-08-05T13:25:18.67Z" },
+ { url = "https://files.pythonhosted.org/packages/8b/83/6d7acb682e98e333bd206e823696c3db55eed2cf9505484f7be3059b1888/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:65bae2a2f028d536706d9e2c24037344154864f84e22aba244f12648b072fb83", size = 12166823, upload-time = "2026-08-05T13:25:20.668Z" },
+ { url = "https://files.pythonhosted.org/packages/84/a9/7433f5599661d5a89eec6c74fd37818a7de87cbe28178ce9ef3591bfbd8b/qdrant_edge_py-0.8.0-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:cceeada8e740796ce07e9239ee5ffe34d6fd0d5a6bad0c0b4a834b93d4abc696", size = 11734612, upload-time = "2026-08-05T13:25:22.537Z" },
+ { url = "https://files.pythonhosted.org/packages/e9/68/a3dfdf201a36828921fba5386e860fca06b906a9c83242c305f47429834a/qdrant_edge_py-0.8.0-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:91b647b128bc79a6c679e7ce8d10a5efdd9dc57268a4ab4eeeb03d84146e71a9", size = 12401117, upload-time = "2026-08-05T13:25:24.438Z" },
+ { url = "https://files.pythonhosted.org/packages/17/65/7066884a033c7926e417b9c5d95e930513151437d455d85aa93add98dbda/qdrant_edge_py-0.8.0-cp310-abi3-win_amd64.whl", hash = "sha256:4e2caeba4db207c6eec6d41ce2f760c32dae36c78761c3640a0e45eadc6d1ec1", size = 11231758, upload-time = "2026-08-05T13:25:26.942Z" },
]
[[package]]
@@ -1527,8 +1529,8 @@ dev = [
requires-dist = [
{ name = "datasets", specifier = ">=4.4.1" },
{ name = "fastembed" },
- { name = "qdrant-client", git = "https://github.com/qdrant/qdrant-client?tag=v1.18.0" },
- { name = "qdrant-edge-py", specifier = "==0.7.2" },
+ { name = "qdrant-client", git = "https://github.com/qdrant/qdrant-client?tag=v1.19.0" },
+ { name = "qdrant-edge-py", specifier = "==0.8.0" },
]
[package.metadata.requires-dev]
diff --git a/automation/snippets/templates/rust/Cargo.lock b/automation/snippets/templates/rust/Cargo.lock
index 48e956404..f94d77271 100644
--- a/automation/snippets/templates/rust/Cargo.lock
+++ b/automation/snippets/templates/rust/Cargo.lock
@@ -61,6 +61,12 @@ version = "0.2.21"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923"
+[[package]]
+name = "allocator-api2"
+version = "0.4.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c880a97d28a3681c0267bd29cff89621202715b065127cd445fa0f0fe0aa2880"
+
[[package]]
name = "android_system_properties"
version = "0.1.5"
@@ -264,28 +270,6 @@ dependencies = [
"windows-sys 0.61.2",
]
-[[package]]
-name = "async-stream"
-version = "0.3.6"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "0b5a71a6f37880a80d1d7f19efd781e4b5de42c88f0722cc13bcb6cc2cfe8476"
-dependencies = [
- "async-stream-impl",
- "futures-core",
- "pin-project-lite",
-]
-
-[[package]]
-name = "async-stream-impl"
-version = "0.3.6"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d"
-dependencies = [
- "proc-macro2",
- "quote",
- "syn",
-]
-
[[package]]
name = "async-task"
version = "4.7.1"
@@ -333,30 +317,26 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8"
[[package]]
-name = "axum"
-version = "0.7.9"
+name = "aws-lc-rs"
+version = "1.17.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f"
+checksum = "00bdb5da18dac48ca2cc7cd4a98e533e8635a58e2361d13a1a4ee3888e0d72f1"
dependencies = [
- "async-trait",
- "axum-core 0.4.5",
- "bytes",
- "futures-util",
- "http",
- "http-body",
- "http-body-util",
- "itoa",
- "matchit 0.7.3",
- "memchr",
- "mime",
- "percent-encoding",
- "pin-project-lite",
- "rustversion",
- "serde",
- "sync_wrapper",
- "tower 0.5.3",
- "tower-layer",
- "tower-service",
+ "aws-lc-sys",
+ "zeroize",
+]
+
+[[package]]
+name = "aws-lc-sys"
+version = "0.43.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "43103168cc76fe62678a375e722fc9cb3a0146159ac5828bc4f0dfd755c2224c"
+dependencies = [
+ "cc",
+ "cmake",
+ "dunce",
+ "fs_extra",
+ "pkg-config",
]
[[package]]
@@ -365,41 +345,21 @@ version = "0.8.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
dependencies = [
- "axum-core 0.5.6",
+ "axum-core",
"bytes",
"futures-util",
"http",
"http-body",
"http-body-util",
"itoa",
- "matchit 0.8.4",
+ "matchit",
"memchr",
"mime",
"percent-encoding",
"pin-project-lite",
"serde_core",
"sync_wrapper",
- "tower 0.5.3",
- "tower-layer",
- "tower-service",
-]
-
-[[package]]
-name = "axum-core"
-version = "0.4.5"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "09f2bd6146b97ae3359fa0cc6d6b376d9539582c7b4220f041a33ec24c226199"
-dependencies = [
- "async-trait",
- "bytes",
- "futures-util",
- "http",
- "http-body",
- "http-body-util",
- "mime",
- "pin-project-lite",
- "rustversion",
- "sync_wrapper",
+ "tower",
"tower-layer",
"tower-service",
]
@@ -536,6 +496,15 @@ dependencies = [
"constant_time_eq",
]
+[[package]]
+name = "blink-alloc"
+version = "0.4.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "ce4c15bad517bc0fb4a44523adf470e2c3eb3a365769327acdba849948ea3705"
+dependencies = [
+ "allocator-api2 0.4.0",
+]
+
[[package]]
name = "block-buffer"
version = "0.12.0"
@@ -686,12 +655,31 @@ dependencies = [
"windows-link 0.2.1",
]
+[[package]]
+name = "cmake"
+version = "0.1.58"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678"
+dependencies = [
+ "cc",
+]
+
[[package]]
name = "colorchoice"
version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570"
+[[package]]
+name = "combine"
+version = "4.6.7"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "ba5a308b75df32fe02788e748662718f03fde005016435c444eea572398219fd"
+dependencies = [
+ "bytes",
+ "memchr",
+]
+
[[package]]
name = "concurrent-queue"
version = "2.5.0"
@@ -767,6 +755,17 @@ dependencies = [
"memchr",
]
+[[package]]
+name = "core_affinity"
+version = "0.8.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "a034b3a7b624016c6e13f5df875747cc25f884156aad2abd12b6c46797971342"
+dependencies = [
+ "libc",
+ "num_cpus",
+ "winapi",
+]
+
[[package]]
name = "cpufeatures"
version = "0.3.0"
@@ -991,6 +990,12 @@ dependencies = [
"litrs",
]
+[[package]]
+name = "dunce"
+version = "1.0.5"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
+
[[package]]
name = "duplicate"
version = "2.0.1"
@@ -1540,7 +1545,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1"
dependencies = [
"ahash",
- "allocator-api2",
+ "allocator-api2 0.2.21",
]
[[package]]
@@ -1549,7 +1554,7 @@ version = "0.15.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1"
dependencies = [
- "allocator-api2",
+ "allocator-api2 0.2.21",
"equivalent",
"foldhash 0.1.5",
]
@@ -1560,11 +1565,17 @@ version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100"
dependencies = [
- "allocator-api2",
+ "allocator-api2 0.2.21",
"equivalent",
"foldhash 0.2.0",
]
+[[package]]
+name = "hashbrown"
+version = "0.17.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a"
+
[[package]]
name = "heapless"
version = "0.8.0"
@@ -1638,6 +1649,12 @@ version = "1.0.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9"
+[[package]]
+name = "humantime"
+version = "2.4.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15"
+
[[package]]
name = "hybrid-array"
version = "0.4.12"
@@ -1684,7 +1701,6 @@ dependencies = [
"tokio",
"tokio-rustls",
"tower-service",
- "webpki-roots",
]
[[package]]
@@ -1717,7 +1733,7 @@ dependencies = [
"libc",
"percent-encoding",
"pin-project-lite",
- "socket2 0.6.3",
+ "socket2",
"tokio",
"tower-service",
"tracing",
@@ -2022,6 +2038,15 @@ dependencies = [
"either",
]
+[[package]]
+name = "itertools"
+version = "0.15.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc"
+dependencies = [
+ "either",
+]
+
[[package]]
name = "itoa"
version = "1.0.17"
@@ -2075,6 +2100,55 @@ dependencies = [
"syn",
]
+[[package]]
+name = "jni"
+version = "0.22.4"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "5efd9a482cf3a427f00d6b35f14332adc7902ce91efb778580e180ff90fa3498"
+dependencies = [
+ "cfg-if",
+ "combine",
+ "jni-macros",
+ "jni-sys",
+ "log",
+ "simd_cesu8",
+ "thiserror 2.0.18",
+ "walkdir",
+ "windows-link 0.2.1",
+]
+
+[[package]]
+name = "jni-macros"
+version = "0.22.4"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "a00109accc170f0bdb141fed3e393c565b6f5e072365c3bd58f5b062591560a3"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "rustc_version",
+ "simd_cesu8",
+ "syn",
+]
+
+[[package]]
+name = "jni-sys"
+version = "0.4.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c6377a88cb3910bee9b0fa88d4f42e1d2da8e79915598f65fb0c7ee14c878af2"
+dependencies = [
+ "jni-sys-macros",
+]
+
+[[package]]
+name = "jni-sys-macros"
+version = "0.4.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264"
+dependencies = [
+ "quote",
+ "syn",
+]
+
[[package]]
name = "jobserver"
version = "0.1.34"
@@ -2191,9 +2265,9 @@ dependencies = [
[[package]]
name = "log"
-version = "0.4.30"
+version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "616ec5685824bcc94416c6d4a7a446eea774a31efd7062c8480ba6fd06d7a6e5"
+checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
[[package]]
name = "lru-slab"
@@ -2203,9 +2277,9 @@ checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154"
[[package]]
name = "lz4_flex"
-version = "0.13.1"
+version = "0.14.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e"
+checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226"
[[package]]
name = "macro_rules_attribute"
@@ -2223,12 +2297,6 @@ version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "670fdfda89751bc4a84ac13eaa63e205cf0fd22b4c9a5fbfa085b63c1f1d3a30"
-[[package]]
-name = "matchit"
-version = "0.7.3"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94"
-
[[package]]
name = "matchit"
version = "0.8.4"
@@ -2770,9 +2838,9 @@ dependencies = [
[[package]]
name = "prost"
-version = "0.13.5"
+version = "0.14.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5"
+checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1"
dependencies = [
"bytes",
"prost-derive",
@@ -2780,12 +2848,12 @@ dependencies = [
[[package]]
name = "prost-derive"
-version = "0.13.5"
+version = "0.14.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
+checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
dependencies = [
"anyhow",
- "itertools",
+ "itertools 0.14.0",
"proc-macro2",
"quote",
"syn",
@@ -2793,17 +2861,17 @@ dependencies = [
[[package]]
name = "prost-types"
-version = "0.13.5"
+version = "0.14.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16"
+checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a"
dependencies = [
"prost",
]
[[package]]
name = "qdrant-client"
-version = "1.18.0"
-source = "git+https://github.com/qdrant/rust-client?branch=master#357dec9e56da4e5afd41645e8c414873a7f8681d"
+version = "1.19.0"
+source = "git+https://github.com/qdrant/rust-client?branch=master#7c838035ae7b9455636dcaca918a55b7d7ca638f"
dependencies = [
"anyhow",
"derive_builder",
@@ -2816,16 +2884,17 @@ dependencies = [
"semver",
"serde",
"serde_json",
- "thiserror 1.0.69",
+ "thiserror 2.0.18",
"tokio",
- "tonic 0.12.3",
+ "tonic",
+ "tonic-prost",
]
[[package]]
name = "qdrant-edge"
-version = "0.7.2"
+version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "b2e5e480105a06c8f1703082758d74a13c98fb7a94df06e12d914460786e55ea"
+checksum = "0b8072302c87506a34bffec9bc16dbdcd36df8ab1321406b6e141530348c7e54"
dependencies = [
"ahash",
"aligned-vec",
@@ -2835,12 +2904,14 @@ dependencies = [
"bincode 1.3.3",
"bitpacking",
"bitvec",
+ "blink-alloc",
"bytemuck",
"byteorder",
"cc",
"cgroups-rs",
"charabia",
"chrono",
+ "core_affinity",
"crc32c",
"data-encoding",
"docopt",
@@ -2854,10 +2925,11 @@ dependencies = [
"geo",
"geohash",
"half 2.7.1",
+ "humantime",
"indexmap 2.13.0",
"integer-encoding",
"io-uring",
- "itertools",
+ "itertools 0.15.0",
"log",
"lz4_flex",
"macro_rules_attribute",
@@ -2869,6 +2941,7 @@ dependencies = [
"num-derive",
"num-traits",
"num_cpus",
+ "once_cell",
"ordered-float 5.3.0",
"parking_lot",
"permutation_iterator",
@@ -2883,7 +2956,6 @@ dependencies = [
"roaring",
"rustix 1.1.4",
"schemars",
- "seahash",
"self_cell",
"semver",
"serde",
@@ -2893,6 +2965,7 @@ dependencies = [
"serde_json",
"serde_variant",
"sha2",
+ "siphasher",
"slab",
"smallvec",
"strum",
@@ -2904,8 +2977,7 @@ dependencies = [
"thread-priority",
"tinyvec",
"tokio",
- "tonic 0.14.6",
- "typed-arena",
+ "tonic",
"uuid",
"validator",
"vaporetto",
@@ -2925,13 +2997,13 @@ dependencies = [
[[package]]
name = "quick_cache"
-version = "0.6.22"
+version = "0.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "d1c821816e9b928e20e92ed59bb3ac4aab321d16ca2316871c9fe7ca739cd477"
+checksum = "403c1a912fec895cafb223201e368234842acb9220aaf08ab042ae89ba5f135c"
dependencies = [
- "ahash",
"equivalent",
- "hashbrown 0.16.1",
+ "foldhash 0.2.0",
+ "hashbrown 0.17.1",
"parking_lot",
]
@@ -2948,7 +3020,7 @@ dependencies = [
"quinn-udp",
"rustc-hash",
"rustls",
- "socket2 0.6.3",
+ "socket2",
"thiserror 2.0.18",
"tokio",
"tracing",
@@ -2961,6 +3033,7 @@ version = "0.11.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098"
dependencies = [
+ "aws-lc-rs",
"bytes",
"getrandom 0.3.4",
"lru-slab",
@@ -2985,7 +3058,7 @@ dependencies = [
"cfg_aliases",
"libc",
"once_cell",
- "socket2 0.6.3",
+ "socket2",
"tracing",
"windows-sys 0.59.0",
]
@@ -3036,8 +3109,6 @@ version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34af8d1a0e25924bc5b7c43c079c942339d8f0a8b57c39049bef581b46327404"
dependencies = [
- "libc",
- "rand_chacha 0.3.1",
"rand_core 0.6.4",
"serde",
]
@@ -3073,16 +3144,6 @@ dependencies = [
"rand_core 0.5.1",
]
-[[package]]
-name = "rand_chacha"
-version = "0.3.1"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88"
-dependencies = [
- "ppv-lite86",
- "rand_core 0.6.4",
-]
-
[[package]]
name = "rand_chacha"
version = "0.9.0"
@@ -3108,7 +3169,6 @@ version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
dependencies = [
- "getrandom 0.2.17",
"serde",
]
@@ -3224,9 +3284,9 @@ checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a"
[[package]]
name = "reqwest"
-version = "0.12.28"
+version = "0.13.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147"
+checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3"
dependencies = [
"base64",
"bytes",
@@ -3246,14 +3306,12 @@ dependencies = [
"quinn",
"rustls",
"rustls-pki-types",
- "serde",
- "serde_json",
- "serde_urlencoded",
+ "rustls-platform-verifier",
"sync_wrapper",
"tokio",
"tokio-rustls",
"tokio-util",
- "tower 0.5.3",
+ "tower",
"tower-http",
"tower-service",
"url",
@@ -3261,7 +3319,6 @@ dependencies = [
"wasm-bindgen-futures",
"wasm-streams",
"web-sys",
- "webpki-roots",
]
[[package]]
@@ -3383,6 +3440,7 @@ version = "0.23.37"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "758025cb5fccfd3bc2fd74708fd4682be41d99e5dff73c377c0646c6012c73a4"
dependencies = [
+ "aws-lc-rs",
"log",
"once_cell",
"ring",
@@ -3404,15 +3462,6 @@ dependencies = [
"security-framework",
]
-[[package]]
-name = "rustls-pemfile"
-version = "2.2.0"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "dce314e5fee3f39953d46bb63bb8a46d40c2f8fb7cc5a3b6cab2bde9721d6e50"
-dependencies = [
- "rustls-pki-types",
-]
-
[[package]]
name = "rustls-pki-types"
version = "1.14.0"
@@ -3424,11 +3473,39 @@ dependencies = [
]
[[package]]
-name = "rustls-webpki"
-version = "0.103.9"
+name = "rustls-platform-verifier"
+version = "0.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "d7df23109aa6c1567d1c575b9952556388da57401e4ace1d15f79eedad0d8f53"
+checksum = "26d1e2536ce4f35f4846aa13bff16bd0ff40157cdb14cc056c7b14ba41233ba0"
dependencies = [
+ "core-foundation",
+ "core-foundation-sys",
+ "jni",
+ "log",
+ "once_cell",
+ "rustls",
+ "rustls-native-certs",
+ "rustls-platform-verifier-android",
+ "rustls-webpki",
+ "security-framework",
+ "security-framework-sys",
+ "webpki-root-certs",
+ "windows-sys 0.61.2",
+]
+
+[[package]]
+name = "rustls-platform-verifier-android"
+version = "0.1.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f"
+
+[[package]]
+name = "rustls-webpki"
+version = "0.103.13"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e"
+dependencies = [
+ "aws-lc-rs",
"ring",
"rustls-pki-types",
"untrusted",
@@ -3499,12 +3576,6 @@ version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
-[[package]]
-name = "seahash"
-version = "4.1.0"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "1c107b6f4780854c8b126e228ea8869f4d7b71260f962fefb57b996b8959ba6b"
-
[[package]]
name = "security-framework"
version = "3.7.0"
@@ -3546,9 +3617,9 @@ checksum = "b12e76d157a900eb52e81bc6e9f3069344290341720e9178cde2407113ac8d89"
[[package]]
name = "semver"
-version = "1.0.27"
+version = "1.0.28"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "d767eb0aabc880b29956c35734170f26ed551a859dbd361d140cdbeca61ab1e2"
+checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd"
dependencies = [
"serde",
"serde_core",
@@ -3629,9 +3700,9 @@ dependencies = [
[[package]]
name = "serde_json"
-version = "1.0.149"
+version = "1.0.151"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86"
+checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
dependencies = [
"indexmap 2.13.0",
"itoa",
@@ -3652,18 +3723,6 @@ dependencies = [
"syn",
]
-[[package]]
-name = "serde_urlencoded"
-version = "0.7.1"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd"
-dependencies = [
- "form_urlencoded",
- "itoa",
- "ryu",
- "serde",
-]
-
[[package]]
name = "serde_variant"
version = "0.1.3"
@@ -3673,6 +3732,12 @@ dependencies = [
"serde",
]
+[[package]]
+name = "sha1_smol"
+version = "1.0.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "bbfa15b3dddfee50a0fff136974b3e1bde555604ba463834a7eb7deb6417705d"
+
[[package]]
name = "sha2"
version = "0.11.0"
@@ -3713,10 +3778,26 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e320a6c5ad31d271ad523dcf3ad13e2767ad8b1cb8f047f75a8aeaf8da139da2"
[[package]]
-name = "siphasher"
-version = "1.0.2"
+name = "simd_cesu8"
+version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "b2aa850e253778c88a04c3d7323b043aeda9d3e30d5971937c1855769763678e"
+checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520"
+dependencies = [
+ "rustc_version",
+ "simdutf8",
+]
+
+[[package]]
+name = "simdutf8"
+version = "0.1.5"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
+
+[[package]]
+name = "siphasher"
+version = "1.0.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8ee5873ec9cce0195efcb7a4e9507a04cd49aec9c83d0389df45b1ef7ba2e649"
[[package]]
name = "slab"
@@ -3748,22 +3829,13 @@ dependencies = [
"qdrant-client",
"qdrant-edge",
"serde_json",
+ "sha2",
"tempfile",
"tokio",
"ureq",
"uuid",
]
-[[package]]
-name = "socket2"
-version = "0.5.10"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678"
-dependencies = [
- "libc",
- "windows-sys 0.52.0",
-]
-
[[package]]
name = "socket2"
version = "0.6.3"
@@ -3864,9 +3936,9 @@ dependencies = [
[[package]]
name = "sysinfo"
-version = "0.38.4"
+version = "0.39.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "92ab6a2f8bfe508deb3c6406578252e491d299cbbf3bc0529ecc3313aee4a52f"
+checksum = "d2071df9448915b71c4fe6d25deaf1c22f12bd234f01540b77312bb8e41361e6"
dependencies = [
"libc",
"memchr",
@@ -4028,7 +4100,7 @@ dependencies = [
"parking_lot",
"pin-project-lite",
"signal-hook-registry",
- "socket2 0.6.3",
+ "socket2",
"tokio-macros",
"windows-sys 0.61.2",
]
@@ -4108,40 +4180,6 @@ dependencies = [
"winnow 1.0.0",
]
-[[package]]
-name = "tonic"
-version = "0.12.3"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "877c5b330756d856ffcc4553ab34a5684481ade925ecc54bcd1bf02b1d0d4d52"
-dependencies = [
- "async-stream",
- "async-trait",
- "axum 0.7.9",
- "base64",
- "bytes",
- "flate2",
- "h2",
- "http",
- "http-body",
- "http-body-util",
- "hyper",
- "hyper-timeout",
- "hyper-util",
- "percent-encoding",
- "pin-project",
- "prost",
- "rustls-native-certs",
- "rustls-pemfile",
- "socket2 0.5.10",
- "tokio",
- "tokio-rustls",
- "tokio-stream",
- "tower 0.4.13",
- "tower-layer",
- "tower-service",
- "tracing",
-]
-
[[package]]
name = "tonic"
version = "0.14.6"
@@ -4149,7 +4187,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef"
dependencies = [
"async-trait",
- "axum 0.8.9",
+ "axum",
"base64",
"bytes",
"flate2",
@@ -4162,35 +4200,27 @@ dependencies = [
"hyper-util",
"percent-encoding",
"pin-project",
- "socket2 0.6.3",
+ "rustls-native-certs",
+ "socket2",
"sync_wrapper",
"tokio",
"tokio-rustls",
"tokio-stream",
- "tower 0.5.3",
+ "tower",
"tower-layer",
"tower-service",
"tracing",
]
[[package]]
-name = "tower"
-version = "0.4.13"
+name = "tonic-prost"
+version = "0.14.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c"
+checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0"
dependencies = [
- "futures-core",
- "futures-util",
- "indexmap 1.9.3",
- "pin-project",
- "pin-project-lite",
- "rand 0.8.5",
- "slab",
- "tokio",
- "tokio-util",
- "tower-layer",
- "tower-service",
- "tracing",
+ "bytes",
+ "prost",
+ "tonic",
]
[[package]]
@@ -4225,7 +4255,7 @@ dependencies = [
"http-body",
"iri-string",
"pin-project-lite",
- "tower 0.5.3",
+ "tower",
"tower-layer",
"tower-service",
]
@@ -4279,12 +4309,6 @@ version = "0.2.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b"
-[[package]]
-name = "typed-arena"
-version = "2.0.2"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "6af6ae20167a9ece4bcb41af5b80f8a1f1df981f6391189ce00fd257af04126a"
-
[[package]]
name = "typeid"
version = "1.0.3"
@@ -4412,6 +4436,7 @@ dependencies = [
"getrandom 0.4.2",
"js-sys",
"serde_core",
+ "sha1_smol",
"wasm-bindgen",
]
@@ -4600,9 +4625,9 @@ dependencies = [
[[package]]
name = "wasm-streams"
-version = "0.4.2"
+version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "15053d8d85c7eccdbefef60f06769760a563c7f0a9d6902a13d35c7800b0ad65"
+checksum = "9d1ec4f6517c9e11ae630e200b2b65d193279042e28edd4a2cda233e46670bbb"
dependencies = [
"futures-util",
"js-sys",
@@ -4643,6 +4668,15 @@ dependencies = [
"wasm-bindgen",
]
+[[package]]
+name = "webpki-root-certs"
+version = "1.0.9"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b"
+dependencies = [
+ "rustls-pki-types",
+]
+
[[package]]
name = "webpki-roots"
version = "1.0.6"
@@ -5215,18 +5249,18 @@ dependencies = [
[[package]]
name = "zerocopy"
-version = "0.8.49"
+version = "0.8.55"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "bce33a6288fa3f072a8c2c7d0f2fdbb90e28298f0135c1f99b96c3db2efcc60b"
+checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb"
dependencies = [
"zerocopy-derive",
]
[[package]]
name = "zerocopy-derive"
-version = "0.8.49"
+version = "0.8.55"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "8fd425244944f4ab65ccff928e7323354c5a018c75838362fdce749dfad2ee1e"
+checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb"
dependencies = [
"proc-macro2",
"quote",
diff --git a/automation/snippets/templates/rust/Cargo.toml b/automation/snippets/templates/rust/Cargo.toml
index 7d93bd658..362816cf6 100644
--- a/automation/snippets/templates/rust/Cargo.toml
+++ b/automation/snippets/templates/rust/Cargo.toml
@@ -5,9 +5,10 @@ edition = "2024"
[dependencies]
anyhow = "1.0.100"
+base64="0.23.1"
chrono = "0.4"
csv = "1.3"
-qdrant-edge = "0.7.2"
+qdrant-edge = "0.8.0"
fs-err = "3"
ordered-float = "5"
qdrant-client = { git = "https://github.com/qdrant/rust-client", branch = "master" }
@@ -15,4 +16,5 @@ serde_json = "1.0.145"
tempfile = "3"
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
ureq = { version = "3", features = ["json"] }
-uuid = { version = "1.18.1", features = ["v4"] }
+uuid = { version = "1.18.1", features = ["v4", "v5"] }
+sha2 = "0.11"
diff --git a/automation/snippets/uv.lock b/automation/snippets/uv.lock
index e27156012..ead3ac215 100644
--- a/automation/snippets/uv.lock
+++ b/automation/snippets/uv.lock
@@ -70,11 +70,11 @@ wheels = [
[[package]]
name = "idna"
-version = "3.11"
+version = "3.15"
source = { registry = "https://pypi.org/simple" }
-sdist = { url = "https://files.pythonhosted.org/packages/6f/6d/0703ccc57f3a7233505399edb88de3cbd678da106337b9fcde432b65ed60/idna-3.11.tar.gz", hash = "sha256:795dafcc9c04ed0c1fb032c2aa73654d8e8c5023a7df64a53f39190ada629902", size = 194582, upload-time = "2025-10-12T14:55:20.501Z" }
+sdist = { url = "https://files.pythonhosted.org/packages/82/77/7b3966d0b9d1d31a36ddf1746926a11dface89a83409bf1483f0237aa758/idna-3.15.tar.gz", hash = "sha256:ca962446ea538f7092a95e057da437618e886f4d349216d2b1e294abfdb65fdc", size = 199245, upload-time = "2026-05-12T22:45:57.011Z" }
wheels = [
- { url = "https://files.pythonhosted.org/packages/0e/61/66938bbb5fc52dbdf84594873d5b51fb1f7c7794e9c0f5bd885f30bc507b/idna-3.11-py3-none-any.whl", hash = "sha256:771a87f49d9defaf64091e6e6fe9c18d4833f140bd19464795bc32d966ca37ea", size = 71008, upload-time = "2025-10-12T14:55:18.883Z" },
+ { url = "https://files.pythonhosted.org/packages/d2/23/408243171aa9aaba178d3e2559159c24c1171a641aa83b67bdd3394ead8e/idna-3.15-py3-none-any.whl", hash = "sha256:048adeaf8c2d788c40fee287673ccaa74c24ffd8dcf09ffa555a2fbb59f10ac8", size = 72340, upload-time = "2026-05-12T22:45:55.733Z" },
]
[[package]]
diff --git a/netlify.toml b/netlify.toml
index 5025939a0..df540c76e 100644
--- a/netlify.toml
+++ b/netlify.toml
@@ -53,6 +53,16 @@ HUGO_PARAMS_onetrustScriptId = "0196246a-3663-7350-9a45-b65f645d6314"
status = 301
force = true
+[[redirects]]
+ from = "/lp/lucene/calendar/"
+ to = "/contact-us/"
+ status = 302
+
+[[redirects]]
+ from = "/lp/lucene/calendar"
+ to = "/contact-us/"
+ status = 302
+
[[redirects]]
from = "/legal/terms_cloud/"
to = "https://qdrant.to/cloud-terms/"
diff --git a/qdrant-landing/.gitignore b/qdrant-landing/.gitignore
index ac85392ea..34ec76b51 100644
--- a/qdrant-landing/.gitignore
+++ b/qdrant-landing/.gitignore
@@ -3,3 +3,4 @@ node_modules
.idea/
.hugo_build.lock
resources/_gen
+assets/jsconfig.json
diff --git a/qdrant-landing/assets/jsconfig.json b/qdrant-landing/assets/jsconfig.json
deleted file mode 100644
index f4e5b9052..000000000
--- a/qdrant-landing/assets/jsconfig.json
+++ /dev/null
@@ -1,10 +0,0 @@
-{
- "compilerOptions": {
- "baseUrl": ".",
- "paths": {
- "*": [
- "../themes/qdrant-2024/assets/*"
- ]
- }
- }
-}
\ No newline at end of file
diff --git a/qdrant-landing/assets/schema/event-schema.json b/qdrant-landing/assets/schema/event-schema.json
new file mode 100644
index 000000000..84aeb7449
--- /dev/null
+++ b/qdrant-landing/assets/schema/event-schema.json
@@ -0,0 +1,51 @@
+{
+ "@type": "Event",
+ "@id": "{{- with .Params.link -}}{{- . -}}{{- else -}}{{- .Permalink -}}{{- end -}}#event",
+ "name": "{{- with .Params.event_name -}}{{- . | htmlEscape -}}{{- else -}}{{- .Title | htmlEscape -}}{{- end -}}",
+ "description": {{ $description := printf "%s" (.Params.description | plainify | replaceRE "(\n)" "" | replaceRE "[^\\w\\s:\\[\\]{}\"]" "" | htmlEscape ) -}}"{{- $description -}}",
+ {{- $image := "" -}}
+ {{- with .Params.social_preview_image -}}{{- $image = . | absURL -}}{{- end -}}
+ {{- if and (not $image) .Params.preview_image -}}{{- $image = .Params.preview_image | absURL -}}{{- end -}}
+ {{- if $image }}
+ "image": ["{{- $image -}}"],
+ {{- end }}
+ "url": "{{- with .Params.link -}}{{- . -}}{{- else -}}{{- .Permalink -}}{{- end -}}",
+ "startDate": "{{- .Params.start -}}"{{- with .Params.end -}},
+ "endDate": "{{- . -}}"{{- end -}},
+ "eventStatus": "{{- with .Params.eventStatus -}}{{- . -}}{{- else -}}https://schema.org/EventScheduled{{- end -}}",
+ {{- $attendanceMode := "https://schema.org/OfflineEventAttendanceMode" -}}
+ {{- if eq (lower (printf "%s" .Params.place)) "online" -}}
+ {{- $attendanceMode = "https://schema.org/OnlineEventAttendanceMode" -}}
+ {{- end -}}
+ {{- with .Params.eventAttendanceMode -}}{{- $attendanceMode = . -}}{{- end }}
+ "eventAttendanceMode": "{{- $attendanceMode -}}",
+ {{- if .Params.location }}
+ "location": {
+ "@type": "Place",
+ "name": "{{- .Params.location.name | htmlEscape -}}"{{- with .Params.location.streetAddress -}},
+ "address": {
+ "@type": "PostalAddress",
+ "streetAddress": "{{- . | htmlEscape -}}",
+ "addressLocality": "{{- $.Params.location.addressLocality | htmlEscape -}}",
+ "addressRegion": "{{- $.Params.location.addressRegion | htmlEscape -}}",
+ "postalCode": "{{- $.Params.location.postalCode | htmlEscape -}}",
+ "addressCountry": "{{- $.Params.location.addressCountry | htmlEscape -}}"
+ }{{- end -}}
+ },
+ {{- else if eq (lower (printf "%s" .Params.place)) "online" }}
+ "location": {
+ "@type": "VirtualLocation"{{- with .Params.link -}},
+ "url": "{{- . -}}"{{- end -}}
+ },
+ {{- else if .Params.place }}
+ "location": {
+ "@type": "Place",
+ "name": "{{- .Params.place | htmlEscape -}}"
+ },
+ {{- end }}
+ "organizer": {
+ "@type": "Organization",
+ "name": "Qdrant",
+ "url": "https://qdrant.tech"
+ }
+}
diff --git a/qdrant-landing/assets/schema/organization-schema.json b/qdrant-landing/assets/schema/organization-schema.json
index aa332167c..87b6cb3e8 100644
--- a/qdrant-landing/assets/schema/organization-schema.json
+++ b/qdrant-landing/assets/schema/organization-schema.json
@@ -12,7 +12,7 @@
"founders": [
{
"@type": "Person",
- "name": "{{ .Site.Params.Author }}"
+ "name": "Andrey Vasnetsov"
}, {
"@type": "Person",
"name": "Andre Zayarni"
diff --git a/qdrant-landing/config.toml b/qdrant-landing/config.toml
index 45d74443f..7611cd8f1 100644
--- a/qdrant-landing/config.toml
+++ b/qdrant-landing/config.toml
@@ -62,12 +62,6 @@ disableKinds = ["taxonomy", "term"]
email = "info@qdrant.tech"
location = "Berlin"
- mailchimp_contact_form = "https://qdrant.to/contact-us"
-
- mailchimp_subscribe = "https://tech.us1.list-manage.com/subscribe/post?u=69617d79374ac6280dd2230b2&id=acb2b876fc"
-
- mailchimp_subscribe_id = "b_69617d79374ac6280dd2230b2_acb2b876fc"
-
prices_contact_form = "https://share-eu1.hsforms.com/1olUNSzuRRWWXxKJevWfxiA2b46ng"
gdpr = "We use cookies to learn more about you. At any time you can delete or block cookies through your browser settings."
@@ -125,10 +119,17 @@ disableKinds = ["taxonomy", "term"]
isHTML = false
rel = "alternate"
+[outputFormats.LLMs]
+ mediaType = "text/plain"
+ baseName = "llms"
+ isPlainText = true
+ isHTML = false
+ notAlternative = true
+
[outputs]
page = ["HTML", "Markdown"]
section = ["HTML", "RSS", "Markdown"]
- home = ["HTML", "RSS"]
+ home = ["HTML", "RSS", "LLMs"]
[services]
[services.googleAnalytics]
diff --git a/qdrant-landing/content/ai-agents/ai-agents-features.md b/qdrant-landing/content/ai-agents/ai-agents-features.md
index f7c0d983a..fd01fd504 100644
--- a/qdrant-landing/content/ai-agents/ai-agents-features.md
+++ b/qdrant-landing/content/ai-agents/ai-agents-features.md
@@ -48,7 +48,7 @@ features:
description: Qdrant’s architecture is optimized for high-throughput embedding processing, minimizing CPU load and preventing performance bottlenecks. This enables AI agents in Agentic RAG workflows to execute complex, multi-step tasks efficiently, ensuring smooth operation even at scale.
link:
text: Distributed Deployment
- url: /documentation/distributed_deployment/
+ url: /documentation/scaling/distributed_deployment/
- id: 4
icon:
src: /icons/outline/speedometer-blue.svg
diff --git a/qdrant-landing/content/articles/agentic-builders-guide.md b/qdrant-landing/content/articles/agentic-builders-guide.md
index 75c46cd04..847a9b2ba 100644
--- a/qdrant-landing/content/articles/agentic-builders-guide.md
+++ b/qdrant-landing/content/articles/agentic-builders-guide.md
@@ -126,7 +126,7 @@ The same agent that speeds through a toy dataset with 10,000 points will become
We’ll talk about three concepts you can take advantage of to improve your scale, but if you want even more information on how to scale, check out this [article](https://qdrant.tech/documentation/database-tutorials/large-scale-search/) on large scale search.
-As your dataset and traffic grow, Qdrant Cloud offers a suite of features to ensure your system can scale effectively. Horizontal scaling is achieved through [sharding](https://qdrant.tech/articles/multitenancy/), which splits your collection across multiple nodes to distribute the load and improve performance. For high availability and fault tolerance, Qdrant supports [replication](https://qdrant.tech/documentation/distributed_deployment/), creating copies of your shards across the cluster.
+As your dataset and traffic grow, Qdrant Cloud offers a suite of features to ensure your system can scale effectively. Horizontal scaling is achieved through [sharding](https://qdrant.tech/articles/multitenancy/), which splits your collection across multiple nodes to distribute the load and improve performance. For high availability and fault tolerance, Qdrant supports [replication](https://qdrant.tech/documentation/scaling/distributed_deployment/), creating copies of your shards across the cluster.
Qdrant provides robust tools for resource and cost optimization. Vector [quantization](https://qdrant.tech/documentation/manage-data/quantization/) compresses your data, significantly reducing its memory footprint and speeding up search.
diff --git a/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md b/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md
new file mode 100644
index 000000000..66b65bb46
--- /dev/null
+++ b/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md
@@ -0,0 +1,230 @@
+---
+title: "What to Check Before Tuning a Qdrant Collection"
+short_description: "Seven collection settings that degrade retrieval without an error, the order to try changes in, and how many labeled queries a gain needs."
+description: "Audit a Qdrant collection: find the settings that degrade retrieval silently, choose the cheapest next change, and size a labeled query set."
+preview_dir: /articles_data/before-tuning-a-qdrant-collection/preview
+social_preview_image: /articles_data/before-tuning-a-qdrant-collection/preview/social_preview.jpg
+weight: -214
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-20T00:00:00+03:00
+draft: false
+keywords:
+ - retrieval tuning
+ - search relevance
+ - nDCG
+ - labeled query set
+ - Qdrant collection audit
+category: search-quality
+---
+
+Before you change a setting, decide what better retrieval means for your workload. The right document at rank one, more candidates for a reranker, lower latency, and a smaller memory footprint each favor different settings, so pick your goal first. If your labeled queries can't detect the improvement you're chasing, you won't be able to tell whether a change helped.
+
+Some settings are there to verify correctness, not to tune performance. If a vector is unindexed, a sparse vector is missing the IDF modifier, or the BM25 average length is wrong, the results are invalid. Any benchmark or comparison you run after that will reflect a broken setup. This article shows you how to check each setting and what the correct state looks like.
+
+## The Retrieval Pipeline You Are Tuning
+
+Every query first retrieves candidates, then ranks them. In dense-only search, one vector search does both. Hybrid search adds a sparse prefetch for exact terms, then fusion combines the dense and sparse candidate lists. A reranker, if present, scores the top candidates again.
+
+
+
+_The hybrid pipeline and the settings each stage owns. Dense-only search uses the dense prefetch path on its own, so `limit` and `hnsw_ef` are its only settings here._
+
+If you run dense-only search and exact keywords are missing from results, hybrid search is the first change to test. [Tuning hybrid search](/articles/how-to-tune-hybrid-search/) covers the request shape, what the second prefetch costs, and how to check that fusion beats either prefetch on your labels.
+
+Before you tune:
+
+1. Check that vectors are indexed and that every field used in a filter has a payload index. [Collection details](/documentation/manage-data/collections/#collection-info) and [payload indexing](/documentation/manage-data/indexing/#payload-index) show what to inspect.
+2. Build a labeled query set and choose a metric that matches the product experience. A labeled query pairs a real user query with the documents that should be returned. [Measuring retrieval relevance](/documentation/improve-search/retrieval-relevance/) walks through the setup.
+
+## The Symptom Tells You Where to Start
+
+Start with the failure mode, not the config reference. The table maps each symptom to the first useful check and the article that covers it.
+
+| What You See | First Check | Read Next |
+|---|---|---|
+| You cannot separate a gain from noise | Build labeled queries, choose a metric, and calculate an interval | This article |
+| Relevant documents do not appear | Measure whether candidate depth is limiting recall | [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) |
+| Keywords, identifiers, SKUs, or error codes do not match | Add a sparse prefetch and measure fusion against each prefetch alone | [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/) |
+| Relevant documents are present but misordered | For hybrid search, tune fusion. If the candidate list needs another ranking stage, test a reranker | [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/), [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) |
+| Results repeat near-duplicates | Test maximal marginal relevance. If chunks from one document fill the page, use grouping | [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) |
+| Search misses its p95 target | Measure the cost of candidate depth before adding another retrieval stage | [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) |
+| The collection no longer fits in RAM | Test memory placement and rescoring | [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/) |
+
+## How to Read These Measurements
+
+The procedure transfers: choose a metric that matches the product experience, compare settings on labeled queries, and validate the winner on fresh queries.
+
+
+
+Qdrant's API and algorithm mechanics carry across collections. The result of a parameter sweep depends on the embedding model, dataset, query mix, filters, index state, shard layout, and deployment. Use each result to choose a test on your own collection, then keep only the settings your labels support.
+
+## Silent Settings Can Break Quality
+
+Check the stages you run before tuning anything else. Each prerequisite has a correct state for a given collection and can fail without an error. Fix them before you benchmark or compare settings, otherwise you are measuring a configuration error, not a trade-off.
+
+### Dense Search and Indexing
+
+**[Vectors are indexed](/documentation/manage-data/collections/#collection-info)** Call `GET /collections/{collection_name}` and compare `indexed_vectors_count` with `points_count`. In a dense-only collection, the counts should match once indexing is complete. In a hybrid collection, where each point has one dense and one sparse vector, `indexed_vectors_count` should be twice `points_count`, because Qdrant counts each vector separately.
+
+If the indexed count is lower, indexing may still be running, may have stopped, or some segments may be smaller than the default `indexing_threshold` of 10,000 KB. See the [indexing optimizer documentation](/documentation/ops-optimization/optimizer/#indexing-optimizer). Qdrant builds an HNSW graph only after a segment reaches `indexing_threshold`. Before then, it searches the segment without HNSW, so changing `hnsw_ef` has no effect.
+
+**[full_scan_threshold](/documentation/manage-data/indexing/#vector-index)** Dense and sparse vectors have separate thresholds in different units, so a value copied between them lands nowhere near the intended size. The dense threshold counts kilobytes of vectors in a segment, 10,000 by default. It sends a search to an exact scan instead of the graph when the segment holds fewer vectors than that, or when a filter matches fewer points than that.
+
+The sparse threshold counts vectors, 5,000 by default, and applies only when a filter is present.
+
+### Sparse Retrieval
+
+These settings apply whether the collection has thousands of documents or billions.
+
+**[Modifier.IDF](/documentation/manage-data/indexing/#idf-modifier)** Use this modifier for sparse vectors from BM25 or miniCOIL. Both leave inverse document frequency (IDF) to Qdrant, which computes it per shard for each query term and weights the term by it. SPLADE already includes corpus-level term weighting, so applying the modifier would count rarity twice.
+
+**[BM25 avg_len](/documentation/search/text-search/full-text-search/#configuring-bm25-parameters)** Set `avg_len` to the average number of tokens in the field after BM25 [stems words and removes stopwords](/documentation/search/text-search/full-text-search/#bm25-text-processing). BM25 uses this value to adjust for document length. Do not estimate it from raw word counts. In the five datasets tested here, the stemmed count was 15% to 43% lower. The correct values ranged from 35.3 to 151.4, compared with the default of 256. Measure it using the same stemmer and stopword settings as the collection.
+
+### Hybrid Search
+
+Fusion placement matters on sharded collections. `score_threshold` is a risk at any scale when a request moves from single-vector retrieval to fusion.
+
+**[Fusion placement](/documentation/search/hybrid-queries/)** At the root of the query, fusion runs once, after every shard returns its candidates. Inside a `prefetch`, fusion runs on each shard. Each shard fuses only its own candidates, and the outer query ranks by those shard-local fused scores. The result changes with shard count and with how points are distributed, and no error tells you it happened. Nested fusion is deliberate when an outer stage rescores its output. On a single-shard collection, both placements produce the same ranking.
+
+**[score_threshold](/documentation/search/search/#filtering-results-by-score)** Use `score_threshold` only when you have a measured minimum acceptance score for the stage that returns results. A threshold copied from dense-only search is unsafe in a root-level RRF or DBSF query. Qdrant compares it with the fused score, not the dense or sparse score. It can silently truncate the result list or return no results. Validate it on labeled queries, or leave it unset.
+
+### Filtered Search
+
+Index every field you filter on. The cost of skipping one grows with collection size and query concurrency.
+
+**[Payload indexes](/documentation/manage-data/indexing/#payload-index)** A healthy collection has a payload index for every field used in its filters. Create these indexes before ingestion. If you add one later, Qdrant does not add the filter-aware HNSW edges automatically. You must [rebuild the HNSW index](/documentation/manage-data/indexing/#rebuild-the-hnsw-index). Qdrant Cloud strict mode rejects queries that filter on unindexed fields. Even with the right indexes, strict filters can reduce recall. [What ACORN fixes, and what fixes ACORN](/articles/filtered-vector-search-acorn/) measures this effect on one million points.
+
+## Change Things in Cost Order
+
+Start with a change that does not rebuild the collection or add a retrieval stage. Move to a higher-cost tier only when the lower-cost options do not address the symptom.
+
+| Tier | What | Applies To | Cost |
+|---|---|---|---|
+| No New Retrieval Work | Fusion method, RRF `k`, weights | Hybrid search | Reorders lists you already retrieved. No rebuild or extra retrieval stage |
+| Expanded Retrieval | `hnsw_ef` | Dense search | Increases search breadth and query time |
+| Expanded Retrieval | Prefetch `limit` | Any pipeline with a downstream stage | Retrieves more candidates, increasing query time |
+| Expanded Retrieval | `full_scan_threshold` | Dense search, especially filtered search | Uses exact scans for larger candidate pools, which can increase query time |
+| A New Stage | Sparse prefetch | Dense-only search | A second index, a second vector per point, and 0.6 to 1.5 ms of query time on one shard |
+| A New Stage | Reranker | Any pipeline | A model call per candidate |
+| Rebuild | Embedding model, `m` | Every collection | Re-indexing the collection. Changing the embedding model also means generating a new vector for every point |
+| Rebuild | Quantization | Collections limited by memory | Re-indexing, plus a compressed copy of every vector. Holding ranking quality then depends on rescoring |
+
+Consider a model-level rebuild only when it addresses a measured constraint, since a new embedding model means re-embedding every point. [How to choose an embedding model](/articles/how-to-choose-an-embedding-model/) covers that decision. When memory is the constraint, a Matryoshka model's [`mrl` parameter](/documentation/inference/matryoshka-models/) shortens the vector itself, which is a different trade from compressing it with quantization.
+
+## Choose a Metric Before You Tune
+
+Choose the metric before you compare settings, because the metric decides the winner. In our testing, `nDCG@10`, `MRR@10`, and `Recall@100` each name a different best setting, and `Recall@100` disagrees with `nDCG@10` on four of five datasets.
+
+**`nDCG@k`** rewards relevant results near the top, gives additional credit when labels are graded, and normalizes each query against a perfect ranking. Use it when rank order among several results matters.
+
+**`MRR@k`** is the mean of one over the rank of the first relevant result. It asks how fast you got to something good. Use it when a query has one right answer.
+
+**`Recall@k`** is the share of all relevant documents that made it into the top k. Use it when you measure a first stage that feeds something else. It is capped per query by the number of relevant documents: a query with 359 relevant documents cannot exceed 0.28 at `Recall@100`, because only 100 can fit. The average across queries can land higher, because queries with fewer relevant documents are not held to that cap. In our testing, one dataset averages 358.9 relevant documents per query, and its best `Recall@100` was 0.3877. Count relevant documents per query before choosing k.
+
+## Make Sure Your Labels Can Detect a Gain
+
+[Retrieval relevance](/documentation/improve-search/retrieval-relevance/) covers building a labeled set. Its size decides whether any retrieval tuning is visible to you at all.
+
+A labeled set is large enough when it can distinguish the improvement you care about from normal query-to-query variation. Size alone will not save an unrepresentative set. Pull queries across the mix your product sees, including its important query types and filters, and spot-check a sample of the labels yourself.
+
+Every check below takes one score per query for each setting you are comparing. Use the Qdrant request your service already sends. The scoring is the same whether your pipeline runs dense-only search, hybrid fusion, or a reranker.
+
+Scoring starts with the metric itself. `dcg` sums graded relevance with a discount that grows with rank. `ndcg_at_k` runs that sum on what came back, then divides it by the same sum over the best ordering the query's labels allow.
+
+```python
+import math
+
+
+def dcg(gains):
+ """Relevance summed with a discount that grows with rank."""
+ return sum(gain / math.log2(rank + 2) for rank, gain in enumerate(gains))
+
+
+def ndcg_at_k(doc_ids, relevance, k=10):
+ """One query's ranking against the best ranking its labels allow."""
+ returned = [relevance.get(doc_id, 0) for doc_id in doc_ids[:k]]
+ ideal = sorted(relevance.values(), reverse=True)[:k]
+ return dcg(returned) / dcg(ideal) if any(ideal) else 0.0
+```
+
+Then run your labeled queries through both settings. You write `search`, which applies one setting to the request your service already sends and returns the points as the server ranked them. Add `with_payload=["doc_id"]` to that request so every point carries the ID your labels use, or read `point.id` if your point IDs are already your document IDs. `score` turns each list into one number, and subtracting the two scores for each query gives the per-query gain.
+
+```python
+# Relevance keyed by the document IDs your labels already use.
+qrels = {"q1": {"doc-41": 1, "doc-77": 2}}
+# Your labeled queries. Each value is what search sends to Qdrant: text or a vector.
+queries = {"q1": [...]}
+# The one parameter under test, in whatever form your search applies it.
+current_setting = {"hnsw_ef": 64}
+candidate_setting = {"hnsw_ef": 256}
+
+
+def search(query_id, query, setting):
+ """You write this: your own Qdrant request, with setting applied.
+
+ Return the points in the order the server ranked them, each carrying doc_id.
+ """
+ raise NotImplementedError
+
+
+def score(queries, qrels, search, setting):
+ """One nDCG@10 per query, for one setting."""
+ return {
+ query_id: ndcg_at_k(
+ [point.payload["doc_id"] for point in search(query_id, query, setting)],
+ qrels.get(query_id, {}),
+ )
+ for query_id, query in queries.items()
+ }
+
+
+candidate = score(queries, qrels, search, candidate_setting)
+current = score(queries, qrels, search, current_setting)
+per_query_gain = [candidate[q] - current[q] for q in sorted(queries)]
+```
+
+The two calls must differ in exactly one setting. Filters, query shape, and candidate limits stay identical. For `MRR@10` and `Recall@100`, [pytrec_eval](https://github.com/cvangysel/pytrec_eval) computes both from the same `qrels`.
+
+Resample the per-query gains with replacement to estimate how much the average gain would move if you had drawn a different set of queries. The resulting 95% interval shows the range consistent with that sampling variation. If the interval includes zero, your labels cannot establish a quality gain.
+
+```python
+import numpy as np
+
+def interval(per_query_gain, resamples=1000, seed=42):
+ """95% interval for the mean per-query gain of one setting over another."""
+ gains = np.asarray(per_query_gain, dtype=float)
+ rng = np.random.default_rng(seed)
+ draws = rng.integers(0, len(gains), size=(resamples, len(gains)))
+ return np.percentile(gains[draws].mean(axis=1), [2.5, 97.5])
+```
+
+The more labeled queries you evaluate, the more precise the measured gain. Across our datasets, the 95% interval typically extended this far above and below the `nDCG@10` gain:
+
+| Labeled Queries | Interval, Either Side of the Gain |
+|---|---|
+| 25 | 0.047 |
+| 50 | 0.035 |
+| 100 | 0.025 |
+| 200 | 0.018 |
+| 300 | 0.015 |
+
+The label count you need depends primarily on effect size and query-to-query variation, not collection size alone.
+
+In our measurements, [fusion settings](/articles/how-to-tune-hybrid-search/) moved `nDCG@10` by 0.012 to 0.038, gains from tuning an already-working collection rather than rebuilding the retrieval pipeline.
+
+Fifty labeled queries were enough for the larger gains: the 0.038 gain had an interval excluding zero in 93% of draws, while gains under 0.02 cleared that bar in 7% to 38%. Treat small movement as unresolved until you have the labels to measure it.
+
+## Check the Winner on Fresh Queries
+
+A setting selected and evaluated on the same queries will look better than it performs on fresh queries. Split the labeled queries in half: select the winner on one half, then measure its gain on the other. We repeated that split 200 times per dataset.
+
+The selected setting usually transfers. Ranking all 30 settings again on the fresh half, our pick typically landed in the top four, and it fell behind the default in 0% to 6% of splits. The gain does shrink: it retained 67% to 95% of what selection reported, so report the number from the fresh queries.
+
+If you compare separately rebuilt indexes, check top-10 agreement across two builds before you treat a small `nDCG@10` difference as a tuning gain. In our clean rebuild test, query sampling moved `nDCG@10` more than graph variation did.
+
+## Start with One Change
+
+Record the current relevance metric and p95 latency for a representative query set. Choose one low-cost change from the symptom table, validate it on fresh queries, and keep it only if the gain survives. Once you have that baseline, [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) shows how to test whether retrieval depth is the constraint.
diff --git a/qdrant-landing/content/articles/binary-quantization.md b/qdrant-landing/content/articles/binary-quantization.md
index 6066ec4fe..849f8280b 100644
--- a/qdrant-landing/content/articles/binary-quantization.md
+++ b/qdrant-landing/content/articles/binary-quantization.md
@@ -158,9 +158,9 @@ client.update_collection(
When setting search parameters, we specify that we want to use `oversampling` and `rescore`. Here is an example snippet:
```python
-client.search(
+client.query_points(
collection_name="{collection_name}",
- query_vector=[0.2, 0.1, 0.9, 0.7, ...],
+ query=[0.2, 0.1, 0.9, 0.7, ...],
search_params=models.SearchParams(
quantization=models.QuantizationSearchParams(
ignore=False,
diff --git a/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md b/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md
new file mode 100644
index 000000000..a3300ef70
--- /dev/null
+++ b/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md
@@ -0,0 +1,254 @@
+---
+title: "Bulk Uploading Data to Qdrant"
+short_description: "Plan bulk uploads in Qdrant at scale: batching, parallelization, sharding, payload indexes, quantization, and on-disk storage."
+description: "Plan bulk uploads in Qdrant: batching, parallelization, sharding, payload indexes, quantization, and on-disk storage."
+preview_dir: /articles_data/bulk-uploads-in-qdrant/preview
+social_preview_image: /articles_data/bulk-uploads-in-qdrant/preview/social_preview.jpg
+weight: 35
+author: John Kupchanko
+author_link: https://github.com/jkupchanko
+keywords:
+ - bulk upload
+ - vector database
+ - batching
+ - quantization
+ - sharding
+category: production-ops
+date: 2026-07-14T00:00:00.000Z
+draft: false
+---
+
+## Why Bulk Uploading Matters
+
+When you start using Qdrant at scale, one of the first challenges you may run into is uploading large amounts of data efficiently. Small uploads are usually straightforward, but bulk ingestion introduces a different set of concerns. As millions of vectors, payloads, and indexes are written into a collection, the system has to manage memory usage, disk writes, background optimization, and search availability at the same time.
+
+If this process is not planned carefully, bulk uploads can create pressure on RAM, slow down ingestion, increase query latency, or cause the optimizer to fall behind. In more constrained environments, large uploads can even lead to out-of-memory issues or unstable performance.
+
+The goal is not simply to upload data as fast as possible. The goal is to upload data in a way that is predictable and safe for the workload you are running. In this guide, we'll walk through best practices for bulk uploads in Qdrant, including batching, parallelization, sharding, payload indexes, and on-disk vector storage.
+
+## Why Vector Type Matters
+
+Before we get into the best practices, it's important to remember that not all vectors behave the same way during ingestion. Dense and sparse vectors use different indexing approaches, which means they can create different performance considerations during bulk uploads.
+
+Let's quickly break down the difference before moving into the recommended upload strategies.
+
+Dense and sparse vectors behave differently during ingestion because they use different indexing paths in Qdrant. Dense vectors rely on HNSW for fast similarity search. During a large upload, the background optimizer builds and updates this index as new segments are written. This can add CPU and memory pressure while uploads are in progress.
+
+Sparse vectors use a separate indexing approach, and the sparse index is updated as points are written. This means sparse vector ingestion should not be treated the same way as dense HNSW indexing.
+
+## Choosing the Right Bulk Upload Strategy
+
+Before we go through the best practices, understand **there is no single configuration that works best for every bulk upload**. The right approach depends on what you are trying to improve: upload speed, memory usage, search availability, or a balance of all three.
+
+The safest approach is to choose the right strategy for the workload instead of relying on one universal setting.
+
+## Option 1: Reduce Memory Pressure
+
+_Dense vectors_
+
+Memory usage can become one of the first bottlenecks during a large upload. Dense vectors are usually fixed-size embeddings, and when millions of them are inserted into a collection, the raw vector data alone can take up a large amount of RAM.
+
+A safer approach is to store dense vectors directly on-disk when the collection is created. This allows incoming vector data to use memmap storage from the beginning, instead of relying on background optimization to move vectors from memory to disk later.
+
+
+
+In Python, you can configure this with `on_disk=True` inside `VectorParams`:
+
+```python
+client.create_collection(
+ collection_name="my_collection",
+ vectors_config=models.VectorParams(
+ size=768,
+ distance=models.Distance.COSINE,
+ on_disk=True,
+ ),
+)
+```
+
+**Best fit:** Large dense vector uploads where raw vector data may put pressure on RAM.
+
+**Watch for:** Search performance may depend more on disk access, especially if the workload needs to read original vectors often. You can usually balance this with quantization at search time, but the important part for bulk uploads is that vector storage is handled safely from the beginning.
+
+## Option 2: Create Payload Indexes (Before Uploading)
+
+_Dense vectors_
+
+Use payload indexes before uploading points when you already know which fields will be used for filtering. This matters because dense vector search often relies on HNSW. When filters are part of the query, Qdrant can use payload indexes to make filtered search more efficient.
+
+If those indexes are created after a large dataset has already been uploaded, filtered search will fall back to slower query-time strategies until the HNSW graph is rebuilt. Rebuilding the graph after the fact is resource-intensive and can take a long time.
+
+
+
+Create the payload index before uploading:
+
+```python
+client.create_payload_index(
+ collection_name="my_collection",
+ field_name="category",
+ field_schema=models.PayloadSchemaType.KEYWORD,
+)
+```
+
+**Best fit:** Workloads that already know which payload fields will be used for filtering, such as category, tenant ID, document type, source, or user ID.
+
+**Watch for:** Payload indexes should be intentional. Indexing fields that are not used for filtering can add extra work without helping the upload or search path.
+
+## Option 3: Quantization to Balance Memory and Search Performance
+
+_Dense vectors_
+
+Storing original vectors on-disk can help reduce memory pressure during large uploads. However, this can also make search more dependent on disk access, especially when Qdrant needs to read the original vectors frequently.
+
+Quantization can help balance this tradeoff. Instead of keeping full-size dense vectors in memory, Qdrant can keep a compressed version available while the original vectors remain on-disk.
+
+
+
+In Python, configure TurboQuant when creating the collection. The `bits` parameter sets the compression level: `BITS4` (the default) stays closest to full precision, while `BITS1` gives the most compression.
+
+```python
+client.create_collection(
+ collection_name="my_collection",
+ vectors_config=models.VectorParams(
+ size=768,
+ distance=models.Distance.COSINE,
+ on_disk=True,
+ ),
+ quantization_config=models.TurboQuantization(
+ turbo=models.TurboQuantQuantizationConfig(
+ always_ram=True,
+ bits=models.TurboQuantBitSize.BITS4,
+ )
+ ),
+)
+```
+
+**Best fit:** Dense vector workloads that need lower memory usage while still keeping search performance practical.
+
+**Watch for:** Quantization can affect precision depending on the workload and configuration. For many use cases, this tradeoff is worth it, but search quality and latency should be tested with real data.
+
+## Option 4: Reduce Sparse Index Memory During Uploads
+
+_Sparse vectors_
+
+For large sparse vector workloads, one option is to store the sparse vector index on-disk. This can help reduce memory usage when the sparse index becomes large.
+
+
+
+Enable on-disk storage for the sparse index:
+
+```python
+client.create_collection(
+ collection_name="my_collection",
+ vectors_config={},
+ sparse_vectors_config={
+ "text": models.SparseVectorParams(
+ index=models.SparseIndexParams(
+ on_disk=True,
+ )
+ )
+ },
+)
+```
+
+**Best fit:** Large sparse vector workloads where the sparse index is putting pressure on memory.
+
+**Watch for:** Storing the sparse index on-disk may slow down search because queries can depend more on disk access. If sparse vector search is latency-sensitive, keeping the sparse index in memory may be better.
+
+## Best Practices for Every Upload
+
+The strategies above depend on your workload, such as vector type, memory limits, and search needs. The following techniques are different. Batching, parallelization, and sharding are not situational choices; they apply to any bulk upload and help improve ingestion throughput and stability regardless of how your collection is configured.
+
+> **Tip:** Connect with `QdrantClient(url, prefer_grpc=True)` for bulk work. gRPC has lower overhead than HTTP and is meaningfully faster for large uploads.
+
+### Batch Your Uploads
+
+_Dense & sparse vectors_
+
+Uploading points one at a time can add unnecessary overhead. Each request has to go through the network, the write path, and internal processing. When this happens millions of times, the upload process can become slower than it needs to be.
+
+A better approach is to upload points in batches. Batching allows Qdrant to process groups of points together instead of handling every point as a separate request.
+
+
+
+Set a batch size when uploading points:
+
+```python
+client.upload_points(
+ collection_name="my_collection",
+ points=points,
+ batch_size=256,
+)
+```
+
+**Best fit:** Large uploads where sending one point per request would create too much request overhead.
+
+**Watch for:** A batch size of 64-256 points is a reasonable starting range. Larger batches can improve throughput but increase memory usage and make retries more expensive if a request fails.
+
+### Parallelize Uploads
+
+_Dense & sparse vectors_
+
+A single upload stream may not fully use the available write capacity of your Qdrant deployment. When uploading a large dataset, you can often improve ingestion throughput by sending multiple batches in parallel.
+
+Parallel uploads allow several workers to upload different parts of the dataset at the same time. This keeps Qdrant's write pipeline active, especially when the collection has multiple shards.
+
+
+
+Note: Parallelism gains are not always linear; in some configurations, 2 workers may perform similarly to 1 before improvements appear at higher counts.
+
+Add parallel workers to the upload:
+
+```python
+client.upload_points(
+ collection_name="my_collection",
+ points=points,
+ batch_size=256,
+ parallel=4,
+)
+```
+
+**Best fit:** Large uploads where one upload worker is not enough to use the available write capacity.
+
+**Watch for:** Too much parallelism can create extra pressure on CPU, memory, disk I/O, and network resources. Start with a smaller number first, then increase based on system behavior.
+
+### Use Multiple Shards for Larger Uploads
+
+_Dense & sparse vectors_
+
+For larger uploads, sharding can help Qdrant process writes in parallel. A collection can be created with more than one shard, and each shard has its own write path. With multiple shards, Qdrant distributes ingestion work across independent write paths.
+
+
+
+Set the shard count when creating the collection:
+
+```python
+client.create_collection(
+ collection_name="my_collection",
+ vectors_config=models.VectorParams(
+ size=768,
+ distance=models.Distance.COSINE,
+ on_disk=True,
+ ),
+ shard_number=2,
+)
+```
+
+**Best fit:** Larger uploads where you want more ingestion parallelism, especially when paired with parallel upload workers.
+
+**Watch for:** More shards are not always better. Each shard adds overhead, so the shard count should match the size of the deployment and the amount of write parallelism you actually need.
+
+## Choosing the Right Mix
+
+
+
+Still deciding exactly what to configure for your workload? [Qdrant's Agent Skills](https://qdrant.tech/documentation/skills/) provide hands-on, scenario-based guidance that walks you through the specific settings for your situation.
+
+## It's Not One-Size-Fits-All
+
+Bulk uploads are not just about sending as much data as possible into Qdrant. As datasets grow, the upload process also needs to account for memory usage, indexing behavior, disk writes, search availability, and overall system stability.
+
+The safest approach is to choose the right strategy for the workload instead of relying on one universal configuration. Dense vectors, sparse vectors, and hybrid setups can all create different performance considerations during ingestion.
+
+> **Tip:** After a large upload, confirm the collection status is green and the optimizers have finished before serving production traffic.
+
+By designing the collection and upload process before ingestion starts, you can make bulk uploads more efficient, more stable, and easier to scale as your dataset grows. To size your deployment, use the [Qdrant sizing calculator](https://sizing.qdrant.tech/).
diff --git a/qdrant-landing/content/articles/candidate-depth.md b/qdrant-landing/content/articles/candidate-depth.md
new file mode 100644
index 000000000..a0bfa4e92
--- /dev/null
+++ b/qdrant-landing/content/articles/candidate-depth.md
@@ -0,0 +1,155 @@
+---
+title: "Candidate Depth: How Much Retrieval Is Enough?"
+short_description: "Raising candidate depth raises the best score a later ranking stage could reach, but default fusion barely used that extra room."
+description: "Set candidate depth and hnsw_ef in Qdrant, measure the gap between your ranking and a perfect one, and balance the trade-offs."
+preview_dir: /articles_data/candidate-depth/preview
+social_preview_image: /articles_data/candidate-depth/preview/social_preview.jpg
+weight: -213
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-21T00:00:00+03:00
+draft: false
+keywords:
+ - candidate depth
+ - hnsw_ef
+ - scalar quantization
+ - memory tiers
+ - HNSW tuning
+category: search-quality
+---
+
+Before you tune candidate depth, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline. Everything below measures against that baseline.
+
+Candidate depth is the number of candidates a retrieval stage passes to a later ranking stage. It matters only when a later stage can use the extra candidates. In hybrid search, every `prefetch` carries its own `limit`, and a [multi-stage query](/documentation/search/hybrid-queries/#multi-stage-queries) that nests one prefetch inside another sets a depth at each level. In dense-only or sparse-only search, it is the number of candidates you pass to a reranker or other downstream stage.
+
+
+
+## The Short Version
+
+1. Test `limit` at 100 and 200 for a downstream ranking stage. Treat those values as a starting point, not a production default: `limit` applies per shard, and a reranker scores every candidate.
+2. Before raising [`hnsw_ef`](/documentation/search/search/#search-api), compare approximate-search recall with an exact search. If recall has plateaued, a larger value adds latency without improving recall.
+3. If RAM is the constraint, test quantization before reducing candidate depth. [Quantization](/documentation/manage-data/quantization/) covers the collection settings.
+
+## More Candidates Can Raise the Best Possible Score
+
+Start by measuring the gap between the candidates you retrieved and the order your pipeline returns them in. Use your [labeled query set](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to score the candidate set as if it were ordered perfectly. That is the best possible score any later ranking of those candidates could reach. Compare it with the current score from the same queries. In these hybrid measurements, the current score is fusion's `nDCG@10` over the same candidates. `nDCG@10` grades the top 10 results and gives more credit to relevant documents near the top.
+
+Suppose a query retrieves three relevant documents, and fusion ranks them 4, 30, and 180. The current score sees only the one at rank 4, since the other two sit outside the top 10 it grades. The best possible score reorders those same candidates and puts all three at the top. No later ranking stage could do better with the candidates that were retrieved.
+
+For hybrid search, score the union of the dense and sparse prefetches. For a single-prefetch pipeline, score the candidates passed to the downstream stage.
+
+Each value is the change in `nDCG@10` from `limit=10` to 500.
+
+| Dataset | Best Possible Change | Current Score Change |
+|---|---|---|
+| SciFact | +0.103 | +0.008 |
+| ArguAna | +0.121 | +0.002 |
+| WANDS | +0.124 | +0.007 |
+| CodeSearchNet | +0.149 | +0.010 |
+| DBPedia-entity | +0.282 | +0.003 |
+
+
+
+_The full sweep behind the table. The best possible score climbs at every depth step on every dataset, while the score fusion returns stays almost flat._
+
+The best possible score change rises with corpus size across these five, from 5,183 documents on SciFact to 100,000 on DBPedia-entity, while the current score change stays flat. Size and domain move together here, so re-measure the gap as your own collection grows.
+
+With Qdrant's default [RRF](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf), the top ranks in each `prefetch` contribute far more to the fused score than the tail. Raising `limit` can add candidates without changing the top 10, or replace a more relevant result. The fused score is not always higher at greater depth: CodeSearchNet peaks at `limit=200` and is lower at 500, and DBPedia-entity peaks at 50. Other fusion methods can rank those candidates differently. [Fusion tuning](/articles/how-to-tune-hybrid-search/) shows how to test them on your labels.
+
+Start `limit` around 100 to 200, then test larger values on your own labels. A [reranker](/articles/when-a-reranker-is-worth-it/) can use the added candidates, and a [Formula Query](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) can rescore those same candidates from payload fields.
+
+Raising `limit` adds retrieval work. If a reranker follows, it also increases the number of candidates the reranker scores. In our single-shard tests, raising `limit` from 10 to 500 increased median latency by 37% to 43%. These results establish the direction, not a portable ratio. Measure the change under your own p95 budget, concurrency, and shard fan-out.
+
+
+
+## Raise `hnsw_ef` Only When Recall Is Still Climbing
+
+For dense vectors, `limit` decides how many candidates the dense stage returns, and `hnsw_ef` decides how wide the HNSW graph traversal searches for them, trading approximate-search recall for latency. Measure `limit` against your labels when a downstream stage can use more candidates, and measure `hnsw_ef` against exact search to see whether the traversal still misses neighbors.
+
+
+
+_`hnsw_ef` widens the set of nodes the search visits, not the number of results. Here the wider walk reaches a neighbor the narrow one missed, and it displaces the weakest result._
+
+Run the same check on your own data, with `limit` set to the value your dense-only stage or dense `prefetch` uses.
+
+```python
+import time
+
+from qdrant_client import QdrantClient, models
+
+client = QdrantClient(
+ url="https://YOUR-CLUSTER.cloud.qdrant.io",
+ api_key="",
+)
+# Your own query vectors, embedded with the model the collection was built with.
+queries = [...]
+# The limit your dense-only stage or dense prefetch uses.
+LIMIT = 100
+
+
+def top_ids(vector, **search_params):
+ return {point.id for point in client.query_points(
+ collection_name="products", query=vector, using="dense",
+ limit=LIMIT, search_params=models.SearchParams(**search_params),
+ ).points}
+
+
+# The full scan is the ground truth, and it runs once: it does not depend on hnsw_ef.
+truth = [top_ids(vector, exact=True) for vector in queries]
+
+for ef in (16, 64, 128, 256, 512):
+ found = 0
+ started = time.perf_counter()
+ for vector, wanted in zip(queries, truth):
+ found += len(top_ids(vector, hnsw_ef=ef) & wanted)
+ elapsed_ms = (time.perf_counter() - started) / len(queries) * 1000
+ print(ef, found / (LIMIT * len(queries)), elapsed_ms)
+```
+
+[`exact=True`](/documentation/search/search/#exact-search) runs a full scan. Both columns below come from that loop against a one-shard SciFact collection, over 50 queries, timed from the client so the network round trip sits inside the number:
+
+| `hnsw_ef` | Recall Against Exact | Milliseconds per Query |
+|---|---|---|
+| 16 | 0.986 | 1.98 |
+| 64 | 0.993 | 1.98 |
+| 128 | 0.999 | 2.18 |
+| 256 | 1.000 | 2.45 |
+| 512 | 1.000 | 2.25 |
+
+On these five datasets, raising `hnsw_ef` through 16, 64, 128, and 512 at depth 200 moved fused `nDCG@10` by at most 0.0022. Relevant-document recall in the candidate union moved by at most 0.0040. Median latency rose between 4% and 49% across the five hybrid requests at prefetch `limit=200`. When the graph is already saturated, the wider search budget is close to pure cost.
+
+On your collection, choose the lowest `hnsw_ef` that reaches your recall target inside your latency budget. When recall is flat from the first value, keep `hnsw_ef` where it is and confirm that Qdrant has built an HNSW graph. Qdrant builds that graph after a segment passes the default `indexing_threshold`, and smaller segments use exhaustive search where `hnsw_ef` has no effect. The [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) show how to confirm the graph exists.
+
+Saturation is a property of your own graph. These collections held at most 100,000 documents, built in one batch, unfiltered and unquantized. We ran the same check on the full 4,635,922-document DBPedia-entity collection, and it returned 0.957 of the exact top 10: about 4% of the true nearest neighbors never came back.
+
+If `hnsw_ef` cannot reach your recall target, [`m`](/documentation/manage-data/indexing/#vector-index) increases the graph's connections and [`ef_construct`](/documentation/manage-data/indexing/#vector-index) broadens the search during graph construction. Both raise the recall the index can achieve, and changing either rebuilds the HNSW index. Filters are the other limit on what the traversal reaches: the [ACORN search algorithm](/documentation/search/search/#acorn-search-algorithm) is disabled by default, and its `enable` flag lets the search explore beyond direct graph neighbors when filters exclude them. [ACORN](/articles/filtered-vector-search-acorn/) can run about two to 10 times slower, so use it when several strict payload filters combine.
+
+## When RAM Is the Constraint
+
+`limit` is a query-time budget. Lowering it cuts retrieval work and the candidates a later stage receives, and it leaves the collection's disk and RAM footprint where it was. Quantization moves that footprint, so test it on your labels before lowering `limit` for memory reasons. [TurboQuant in Qdrant](/articles/turboquant-quantization/) compares the storage classes.
+
+Int8 scalar quantization stores a compressed copy at one-quarter the size of the float32 vectors. We rebuilt SciFact and DBPedia-entity with it to measure dense top-10 agreement and the effect on the final hybrid result.
+
+| Setting | Dense Top-10 Agreement with Unquantized | Fused `nDCG@10` Change |
+|---|---|---|
+| No rescoring | 0.984 | -0.0001 to +0.0000 |
+| `rescore=True` | 0.997 to 1.000 | -0.0001 to +0.0000 |
+| `rescore=True`, `oversampling=4` | 0.998 to 1.000 | +0.0000 to +0.0001 |
+
+Quantization does reorder the candidate list: without rescoring, 1.6% of the dense prefetch's top 10 moves, though almost none of that reached our fused results, because the default RRF fusion used ranks. [`rescore`](/documentation/manage-data/quantization/#searching-with-quantization) rescores the shortlist with the original vectors, [`oversampling`](/documentation/manage-data/quantization/#searching-with-quantization) fetches extra compressed candidates for that step to choose from, and on SciFact rescoring recovered the unquantized top 10.
+
+This measurement covers int8 scalar quantization on one shard at 5,000 and 100,000 documents. Binary quantization is a far more aggressive trade and we did not test it here.
+
+Compare quantization with cutting `limit`. Dropping depth from 500 to 10 removed 27% to 30% of median latency in our runs and left the footprint where it was. Int8 quantization stored the vectors at one-quarter the size and moved fused `nDCG@10` by at most 0.0001 in either direction. Of the two, quantization is the one that shrinks what the vectors need in RAM.
+
+Once the collection outgrows RAM, the question stops being how many candidates to fetch and becomes which structures stay resident. [Memory placement and rescoring](/articles/when-your-collection-outgrows-ram/) measures that boundary on 4.6 million vectors and explains the placement rules.
+
+## What to Tune Next
+
+The gap between the best possible score and the current score tells you whether the next experiment should focus on ranking or retrieval. A large gap means relevant candidates are present but not ranked highly enough. In hybrid search, test fusion settings; in any pipeline with a downstream stage, test whether a reranker can recover the gap. A small gap means ranking is already close to the best the candidate set allows, so improve the candidates instead.
+
+Next, if you use hybrid search, [tune fusion over the candidates you already retrieve](/articles/how-to-tune-hybrid-search/).
diff --git a/qdrant-landing/content/articles/data-privacy.md b/qdrant-landing/content/articles/data-privacy.md
index 4269afe85..0145ee407 100644
--- a/qdrant-landing/content/articles/data-privacy.md
+++ b/qdrant-landing/content/articles/data-privacy.md
@@ -18,7 +18,7 @@ keywords: # Keywords for SEO
category: production-ops
---
-Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures become critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as "[embedding inversion attack](https://arxiv.org/abs/2004.00053)," where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves.
+Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures are critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as "[embedding inversion attack](https://arxiv.org/abs/2004.00053)," where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves.
Strict compliance regulations govern data stored in vector databases across various industries. For instance, healthcare must comply with HIPAA, which dictates how protected health information (PHI) is stored, transmitted, and secured. Similarly, the financial services industry follows PCI DSS to safeguard sensitive financial data. These regulations require developers to ensure data storage and transmission comply with industry-specific legal frameworks across different regions. **As a result, features that enable data privacy, security and sovereignty are deciding factors when choosing the right vector database.**
@@ -43,27 +43,55 @@ The primary challenge with static API keys is their all-or-nothing access, inade
One of the cornerstones of our design choices at Qdrant has been the focus on security features. We have built in a range of features keeping the enterprise user in mind, which allow building of granular access control on a fully data sovereign architecture.
-A Qdrant instance is unsecured by default. However, when you are ready to deploy in production, Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture.
+Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture.
-### API Keys and TLS Encryption
+> Self-hosted open source deployments are not secure by default and are not production-ready. Qdrant Cloud deployments are always secure and production-ready.
-For simpler use cases, Qdrant offers API key-based authentication. This includes both regular API keys and read-only API keys. Regular API keys grant full access to read, write, and delete operations, while read-only keys restrict access to data retrieval operations only, preventing write actions.
+Qdrant Cloud has two independent access-control systems, and it helps to keep them separate from the outset:
-On Qdrant Cloud, you can create API keys using the [Cloud Dashboard](https://qdrant.to/cloud). This allows you to generate API keys that give you access to a single node or cluster, or multiple clusters. You can read the steps to do so [here](/documentation/cloud/authentication/).
+- Cloud Access Control
+- Database Access Control
-
+### Cloud RBAC
+
+[Cloud RBAC](https://qdrant.tech/documentation/cloud-rbac/) governs what a user can do inside the Qdrant cloud console and account: managing clusters, billing, identity and access management, Hybrid Cloud, and account settings.
+
+Qdrant Cloud includes some built-in roles for common use-cases.
+A Role contains a set of permissions that define the ability to perform or control specific actions in Qdrant Cloud.
+Under Access Management in the Cloud Dashboard, you can also create custom roles with granular permissions, invite users, and assign roles to them.
+
+Note that current permissions control access to ALL clusters. Per Cluster permissions will be in a future release.
+
+
+
+The keys associated with this layer are Cloud Management Keys, which authenticate to the Qdrant Cloud API.
+
+
+
+### Database API Keys
+
+Database API Keys enable [Database Access Control](https://qdrant.tech/documentation/cloud/authentication/). They are used to read and write data inside your Qdrant collections.
+Qdrant supports three types of API key:
+
+1. **Admin API Key**: grants full access to all operations and collections.
+2. **Read-Only API Key**: grants read-only access to all operations and collections, suitable for services or users that only need to query data.
+3. **Granular Access API Key**: assigns read or write permissions to the whole cluster or on individual collections. These keys are built on the JSON Web Token (JWT) standard and are covered in detail in the sections that follow.
+
+On Qdrant Cloud, you create granular access keys from the API Keys section of a cluster's detail page. Each key is scoped to the single cluster it was created in. You can optionally grant per-collection access levels, so a single key can grant read-write on some collections and read-only on others.
+
+
For on-premise or local deployments, you'll need to configure API key authentication. This involves specifying a key in either the Qdrant configuration file or as an environment variable. This ensures that all requests to the server must include a valid API key sent in the header.
-When using the simple API key-based authentication, you should also turn on TLS encryption. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](/documentation/security/#tls) in the configuration.
+When using the simple API key-based authentication on your self-hosted deployment, you should also turn on **TLS encryption**. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](/documentation/security/#tls) in the configuration.
-API authentication, coupled with TLS encryption, offers a first layer of security for your Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs).
+API authentication, coupled with TLS encryption, offers a first layer of security for your self-hosted Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs).
-### JWT on Qdrant
+#### JWT on Qdrant
JSON Web Tokens (JWTs) are a compact, URL-safe, and stateless means of representing _claims_ to be transferred between two parties. These claims are encoded as a JSON object and are cryptographically signed.
-JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity.
+A JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity.
In Qdrant, JWT forms the foundation through which powerful access controls can be built. Let’s understand how.
@@ -101,16 +129,16 @@ qdrant_client = QdrantClient(
search_vector = [0.1, 0.2, 0.3, 0.4]
# Example similarity search request
-response = qdrant_client.search(
+response = qdrant_client.query_points(
collection_name="demo_collection",
- query_vector=search_vector,
+ query=search_vector,
limit=5 # Number of results to retrieve
)
```
For convenience, we have added a JWT generation tool in the Qdrant Web UI, which is present under the 🔑 tab. For your local deployments, you will find it at [http://localhost:6333/dashboard#/jwt](http://localhost:6333/dashboard#/jwt).
-### Payload Configuration
+#### Payload Configuration
There are several different options (claims) you can use in the JWT payload that help control access and functionality. Let’s look at them one by one.
@@ -147,19 +175,17 @@ Suppose you have a ‘users’ collection and have defined specific roles for ea
"matches": [
{ "key": "username", "value": "john" },
{ "key": "role", "value": "developer" }
- ],
- },
+ ]
+ }
}
```
-
-
Now, if you ever want to revoke access for a user, simply change the value of their role. All future requests will be invalid using a token payload of the above type.
By combining the claims, you can fully customize the access level that a user or a role has within the vector store.
-### Creating Role-Based Access Control (RBAC) Using JWT
+#### Creating Role-Based Access Control (RBAC) for Self-Hosted Instances Using JWT
As we saw above, JWT claims create powerful levers through which you can create granular access control on Qdrant. Let’s bring it all together and understand how it helps you create Role-Based Access Control (RBAC).
@@ -194,8 +220,9 @@ In such an application, an example JWT payload for a customer support representa
}
],
"value_exists": {
- "collection": "departments",
+ "collection": "employees",
"matches": [
+ { "key": "username", "value": "john" },
{ "key": "department", "value": "support" }
]
}
diff --git a/qdrant-landing/content/articles/dedicated-service.md b/qdrant-landing/content/articles/dedicated-service.md
index 8a7eb271b..7a8ad2ee0 100644
--- a/qdrant-landing/content/articles/dedicated-service.md
+++ b/qdrant-landing/content/articles/dedicated-service.md
@@ -1,6 +1,6 @@
---
title: "Do You Need Dedicated Vector Search?"
-short_description: "Why vector search requires to be a dedicated service."
+short_description: "Why vector search needs to be a dedicated service."
description: "Why vector search requires a dedicated service."
social_preview_image: /articles_data/dedicated-service/preview/social_preview.jpg
small_preview_image: /articles_data/dedicated-service/preview/icon.svg
@@ -16,6 +16,13 @@ keywords:
- best practices
- anti-patterns
category: core-concepts
+toc_titles:
+ each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database: "Every DB Will Introduce Vectors"
+ having-a-dedicated-vector-database-requires-duplication-of-data: "Data Duplication"
+ having-a-dedicated-vector-database-requires-complex-data-synchronization: "Data Synchronization"
+ you-have-to-pay-for-a-vector-service-uptime-and-data-transfer-of-both-solutions: "Uptime and Transfer Cost"
+ what-is-more-seamless-than-your-current-database-adding-vector-search-capability: "Seamless Integration"
+ databases-can-support-rag-use-case-end-to-end: "End-to-End RAG Support"
---
@@ -27,13 +34,13 @@ Some say storing them in a specialized engine (aka vector database) is better. O
Here are [just](https://nextword.substack.com/p/vector-database-is-not-a-separate) a [few](https://stackoverflow.blog/2023/09/20/do-you-need-a-specialized-vector-database-to-implement-vector-search-well/) of [them](https://www.singlestore.com/blog/why-your-vector-database-should-not-be-a-vector-database/).
-This article presents our vision and arguments on the topic .
+This article presents our vision and arguments on the topic.
We will:
-1. Explain why and when you actually need a dedicated vector solution
+1. Explain why and when you actually need a dedicated vector solution.
2. Debunk some ungrounded claims and anti-patterns to be avoided when building a vector search system.
-A table of contents:
+Here is a list of claims we will respond to:
* *Each database vendor will sooner or later introduce vector capabilities...* [[click](#each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database)]
* *Having a dedicated vector database requires duplication of data.* [[click](#having-a-dedicated-vector-database-requires-duplication-of-data)]
@@ -45,13 +52,13 @@ A table of contents:
## Responding to claims
-###### Each database vendor will sooner or later introduce vector capabilities. That will make every database a Vector Database.
+### Each database vendor will sooner or later introduce vector capabilities. That will make every database a Vector Database.
The origins of this misconception lie in the careless use of the term Vector *Database*.
When we think of a *database*, we subconsciously envision a relational database like Postgres or MySQL.
Or, more scientifically, a service built on ACID principles that provides transactions, strong consistency guarantees, and atomicity.
-The majority of Vector Database are not *databases* in this sense.
+The majority of Vector Databases are not *databases* in this sense.
It is more accurate to call them *search engines*, but unfortunately, the marketing term *vector database* has already stuck, and it is unlikely to change.
@@ -70,8 +77,9 @@ What types of properties do search engines prioritize?
Those priorities lead to different architectural decisions that are not reproducible in a general-purpose database, even if it has vector index support.
+This is why adding vector search capabilities to an existing database does not automatically turn it into a vector database. For example, pgvector allows PostgreSQL to store and query embeddings while preserving the benefits of a relational database. However, it also inherits PostgreSQL’s underlying architecture, which was designed for transactional workloads rather than large-scale similarity search. This creates tradeoffs around scalability, indexing, filtering, and hybrid search that become increasingly important as vector workloads grow. For a deeper analysis, see our blog post on the [tradeoffs of using pgvector](https://qdrant.tech/blog/pgvector-tradeoffs/).
-###### Having a dedicated vector database requires duplication of data.
+### Having a dedicated vector database requires duplication of data.
By their very nature, vector embeddings are derivatives of the primary source data.
@@ -84,12 +92,12 @@ In systems where vector embeddings are fused with the primary data source, it is
As a result, even if you want to use a single database for storing all kinds of data, you would still need to duplicate data internally.
-###### Having a dedicated vector database requires complex data synchronization.
+### Having a dedicated vector database requires complex data synchronization.
Most production systems prefer to isolate different types of workloads into separate services.
In many cases, those isolated services are not even related to search use cases.
-For example, databases for analytics and one for serving can be updated from the same source.
+For example, databases for analytics and one for serving can be updated from the same source.
Yet they can store and organize the data in a way that is optimal for their typical workloads.
Search engines are usually isolated for the same reason: you want to avoid creating a noisy neighbor problem and compromise the performance of your main database.
@@ -102,14 +110,14 @@ You can probably use the smallest free tier of any cloud provider to host it.
But if we want to use this database for vector search, 1 million OpenAI `text-embedding-ada-002` embeddings will take **~6GB of RAM** (sic!).
As you can see, the vector search use case completely overwhelmed the main database resource requirements.
-In practice, this means that your main database becomes burdened with high memory requirements and can not scale efficiently, limited by the size of a single machine.
+In practice, this means that your main database becomes burdened with high memory requirements and cannot scale efficiently, limited by the size of a single machine.
Fortunately, the data synchronization problem is not new and definitely not unique to vector search.
There are many well-known solutions, starting with message queues and ending with specialized ETL tools.
-For example, we recently released our [integration with Airbyte](/documentation/data-management/airbyte/), allowing you to synchronize data from various sources into Qdrant incrementally.
+For example, we released our [integration with Airbyte](/documentation/data-management/airbyte/), allowing you to synchronize data from various sources into Qdrant incrementally.
-###### You have to pay for a vector service uptime and data transfer of both solutions.
+### You have to pay for a vector service uptime and data transfer of both solutions.
In the open-source world, you pay for the resources you use, not the number of different databases you run.
Resources depend more on the optimal solution for each use case.
@@ -119,7 +127,7 @@ For instance, Qdrant implements a number of [quantization techniques](/documenta
In terms of data transfer costs, on most cloud providers, network use within a region is usually free. As long as you put the original source data and the vector store in the same region, there are no added data transfer costs.
-###### What is more seamless than your current database adding vector search capability?
+### What is more seamless than your current database adding vector search capability?
In contrast to the short-term attractiveness of integrated solutions, dedicated search engines propose flexibility and a modular approach.
You don't need to update the whole production database each time some of the vector plugins are updated.
@@ -136,17 +144,19 @@ In those situations, it is much easier to maintain a dedicated search engine for
Finally, the vector capabilities of the all-in-one database are tied to the development and release cycle of the entire stack.
Their long history of use also means that they need to pay a high price for backward compatibility.
-###### Databases can support RAG use-case end-to-end.
+### Databases can support RAG use-case end-to-end.
Putting aside performance and scalability questions, the whole discussion about implementing RAG in the DBs assumes that the only detail missing in traditional databases is the vector index and the ability to make fast ANN queries.
In fact, the current capabilities of vector search have only scratched the surface of what is possible.
-For example, in our recent article, we discuss the possibility of building an [exploration API](/articles/vector-similarity-beyond-search/) to fuel the discovery process - an alternative to kNN search, where you don’t even know what exactly you are looking for.
+For example, in this article, we discuss building an [exploration API](/articles/vector-similarity-beyond-search/) to fuel the discovery process, an alternative to kNN search, where you don’t even know what exactly you are looking for.
## Summary
-Ultimately, you do not need a vector database if you are looking for a simple vector search functionality with a small amount of data. We genuinely recommend starting with whatever you already have in your stack to prototype. But you need one if you are looking to do more out of it, and it is the central functionality of your application. It is just like using a multi-tool to make something quick or using a dedicated instrument highly optimized for the use case.
+Ultimately, you do not need a vector database if you are looking for a simple vector search functionality with a small amount of data. We genuinely recommend starting with whatever you already have in your stack to prototype. But you need one if you are looking to do more out of it, and it is the central functionality of your application. It is just like using a multi-tool to make something quick or using a dedicated instrument highly optimized for the use case.
-Large-scale production systems usually consist of different specialized services and storage types for good reasons since it is one of the best practices of modern software architecture. Comparable to the orchestration of independent building blocks in a microservice architecture.
+When vector search becomes a core workload, a dedicated service can provide significant performance and efficiency advantages. As an example, Qdrant outperformed Elastic's DiskBBQ at 2x throughput, half the latency, and 1/3 the compute requirements. Read more about the benchmark [here](https://qdrant.tech/blog/benchmark-elastic-diskbbq/).
+
+Large-scale production systems usually consist of different specialized services and storage types for good reasons, since it is one of the best practices of modern software architecture. This is comparable to the orchestration of independent building blocks in a microservice architecture.
When you stuff the database with a vector index, you compromise both the performance and scalability of the main database and the vector search capabilities.
There is no one-size-fits-all approach that would not compromise on performance or flexibility.
diff --git a/qdrant-landing/content/articles/fastembed.md b/qdrant-landing/content/articles/fastembed.md
index 215ac9299..0818ca76b 100644
--- a/qdrant-landing/content/articles/fastembed.md
+++ b/qdrant-landing/content/articles/fastembed.md
@@ -6,9 +6,10 @@ social_preview_image: /articles_data/fastembed/preview/social_preview.jpg
small_preview_image: /articles_data/fastembed/preview/lightning.svg
preview_dir: /articles_data/fastembed/preview
weight: 10
-author: Nirant Kasliwal
-author_link: https://nirantk.com/about/
-date: 2023-10-18T10:00:00+03:00
+author: Nirant Kasliwal and Manas Chopra
+author_link: https://nirantk.com/about/
+date: 2026-07-30T10:00:00+03:00
+featured: false
draft: false
keywords:
- vector search
@@ -32,28 +33,37 @@ To tackle these problems we built a small library focused on the task of quickly
## Quick Embedding Text Document Example
-Here is an example of how simple we have made embedding text documents:
+Here is an example of how simple we have made embedding text documents. First, install FastEmbed:
```python
-documents: List[str] = [
- "Hello, World!",
- "fastembed is supported by and maintained by Qdrant."
-]
-embedding_model = DefaultEmbedding()
-embeddings: List[np.ndarray] = list(embedding_model.embed(documents))
+pip install fastembed
```
-These 3 lines of code do a lot of heavy lifting for you: They download the quantized model, load it using ONNXRuntime, and then run a batched embedding creation of your documents.
+Then, generate embeddings for a list of documents:
+
+```python
+from typing import List
+from fastembed import TextEmbedding
+
+documents: List[str] = [
+ "Hello, World!",
+ "fastembed is supported by and maintained by Qdrant."
+]
+embedding_model = TextEmbedding()
+embeddings: List[np.ndarray] = list(embedding_model.embed(documents))
+```
+
+These last 3 lines of code do a lot of heavy lifting for you: They download the quantized model, load it using ONNXRuntime, and then run a batched embedding creation of your documents.
### Code Walkthrough
Let’s delve into a more advanced example code snippet line-by-line:
```python
-from fastembed.embedding import DefaultEmbedding
+from fastembed import TextEmbedding
```
-Here, we import the FlagEmbedding class from FastEmbed and alias it as Embedding. This is the core class responsible for generating embeddings based on your chosen text model. This is also the class which you can import directly as DefaultEmbedding which is [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)
+Here, we import the `TextEmbedding` class from FastEmbed. This is the core class responsible for generating embeddings based on your chosen text model. By default, it loads [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)
```python
documents: List[str] = [
@@ -73,7 +83,7 @@ The use of text prefixes like “query” and “passage” isn’t merely synta
Next, we initialize the Embedding model with the default model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5).
```python
-embedding_model = DefaultEmbedding()
+embedding_model = TextEmbedding()
```
The default model and several other models have a context window of a maximum of 512 tokens. This maximum limit comes from the embedding model training and design itself. If you'd like to embed sequences larger than that, we'd recommend using some pooling strategy to get a single vector out of the sequence. For example, you can use the mean of the embeddings of different chunks of a document. This is also what the [SBERT Paper recommends](https://lilianweng.github.io/posts/2021-05-31-contrastive/#sentence-bert)
@@ -90,15 +100,16 @@ The `embed()` method returns a list of NumPy arrays, each corresponding to the
You can easily parse these NumPy arrays for any downstream application—be it clustering, similarity comparison, or feeding them into a machine learning model for further analysis.
-## 3 Key Features of FastEmbed
+## Why FastEmbed is Useful
FastEmbed is built for inference speed, without sacrificing (too much) performance:
-1. 50% faster than PyTorch Transformers
-2. Better performance than Sentence Transformers and OpenAI Ada-002
-3. Cosine similarity of quantized and original model vectors is 0.92
+- **Light**: Unlike other inference frameworks, such as PyTorch, FastEmbed requires very little in the way of external dependencies. Because it uses the ONNX Runtime, it’s perfect for serverless environments like AWS Lambda.
+- **Fast**: By using ONNX, FastEmbed ensures high-performance inference across various hardware platforms.
+- **Accurate**: FastEmbed aims for better accuracy and recall than models like OpenAI’s `Ada-002`. It always uses models which demonstrate strong results on the MTEB leaderboard.
+- **Support**: FastEmbed supports a wide range of models — dense, sparse, and multi-vector, including multilingual ones — to meet diverse use case needs.
-We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we've chosen that for comparison:
+We use `BAAI/bge-small-en-v1.5` as our default `TextEmbedding` model:

@@ -106,58 +117,49 @@ We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we've chosen that
**Quantized Models**: We quantize the models for CPU (and Mac Metal) – giving you the best buck for your compute model. Our default model is so small, you can run this in AWS Lambda if you’d like!
-Shout out to Huggingface's [Optimum](https://github.com/huggingface/optimum) – which made it easier to quantize models.
+Shout out to Huggingface's [Optimum](https://github.com/huggingface/optimum) – which made it easier to quantize models.
-**Reduced Installation Time**:
+**Light on Dependencies**:
-FastEmbed sets itself apart by maintaining a low minimum RAM/Disk usage.
+FastEmbed sets itself apart by maintaining a low minimum RAM/Disk usage. Unlike other inference frameworks, such as PyTorch, it requires very little in the way of external dependencies, and there’s no requirement for CUDA drivers to run on CPU.
-It’s designed to be agile and fast, useful for businesses looking to integrate text embedding for production usage. For FastEmbed, the list of dependencies is refreshingly brief:
+This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups, while still remaining agile and fast enough for production use.
-> - onnx: Version ^1.11 – We’ll try to drop this also in the future if we can!
-> - onnxruntime: Version ^1.15
-> - tqdm: Version ^4.65 – used only at Download
-> - requests: Version ^2.31 – used only at Download
-> - tokenizers: Version ^0.13
-
-This minimized list serves two purposes. First, it significantly reduces the installation time, allowing for quicker deployments. Second, it limits the amount of disk space required, making it a viable option even for environments with storage limitations.
-
-Notably absent from the dependency list are bulky libraries like PyTorch, and there’s no requirement for CUDA drivers. This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups.
-
-**ONNXRuntime**: The ONNXRuntime gives us the ability to support multiple providers. The quantization we do is limited for CPU (Intel), but we intend to support GPU versions of the same in the future as well. This allows for greater customization and optimization, further aligning with your specific performance and computational requirements.
+**ONNXRuntime and GPU Support**: The ONNXRuntime gives us the ability to support multiple providers. FastEmbed's default quantization targets CPU, but GPU acceleration is also available: install `fastembed-gpu` and set `cuda=True` (with `device_ids` to spread work across multiple GPUs) to run inference on GPU instead of CPU. FastEmbed also supports parallelizing inference across multiple CPU workers with the `parallel` parameter, and lazy model loading with `lazy_load`, to optimize throughput for large-scale indexing pipelines.
## Current Models
-We’ve started with a small set of supported models:
+FastEmbed has grown well beyond dense text embeddings. Today it supports:
-All the models we support are [quantized](https://pytorch.org/docs/stable/quantization.html) to enable even faster computation!
+- **Dense embeddings** – the default `TextEmbedding` model used throughout this article (e.g. `BAAI/bge-small-en-v1.5`, multilingual-e5, nomic-embed-text-v2-moe)
+- **Sparse embeddings** – `SparseTextEmbedding` models including BM25, SPLADE, and miniCOIL for exact keyword-style retrieval
+- **Multi-vector embeddings** – `LateInteractionTextEmbedding` models including ColBERT, ideal for rescoring and small-scale retrieval
+- **Image embeddings** – `ImageEmbedding` models including CLIP variants for visual and multimodal search
+- **Rerankers** – `TextCrossEncoder` cross-encoders to re-rank top-K results (e.g. ms-marco-MiniLM)
+- **Postprocessing** – MUVERA, for compressing multi-vector embeddings into single fixed-size vectors for fast first-stage search
+
+Most of the models we support are [quantized](https://pytorch.org/docs/stable/quantization.html) to enable even faster computation!
If you're using FastEmbed and you've got ideas or need certain features, feel free to let us know. Just drop an issue on our GitHub page. That's where we look first when we're deciding what to work on next. Here's where you can do it: [FastEmbed GitHub Issues](https://github.com/qdrant/fastembed/issues).
-When it comes to FastEmbed's DefaultEmbedding model, we're committed to supporting the best Open Source models.
+When it comes to FastEmbed's default `TextEmbedding` model, we're committed to supporting the best Open Source models.
-If anything changes, you'll see a new version number pop up, like going from 0.0.6 to 0.1. So, it's a good idea to lock in the FastEmbed version you're using to avoid surprises.
+If anything changes, you'll see a new version number pop up. So, it's a good idea to lock in the FastEmbed version you're using to avoid surprises.
## Using FastEmbed with Qdrant
Qdrant is a Vector Store, offering comprehensive, efficient, and scalable [enterprise solutions](https://qdrant.tech/enterprise-solutions/) for modern machine learning and AI applications. Whether you are dealing with billions of data points, require a low latency performant [vector database solution](https://qdrant.tech/qdrant-vector-database/), or specialized quantization methods – [Qdrant is engineered](/documentation/overview/) to meet those demands head-on.
-The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the DefaultEmbedding and use that with Qdrant.
+The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the default `TextEmbedding` model and use that with Qdrant.
Below is a detailed guide on how to get started with FastEmbed in conjunction with Qdrant.
### Step 1: Installation
-Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip:
+Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip. Wrap the package name in quotes so shells like zsh don't try to expand the brackets:
-```
-pip install qdrant-client[fastembed]
-```
-
-For those using zsh as their shell, you might encounter syntax issues. In such cases, wrap the package name in quotes:
-
-```
-pip install 'qdrant-client[fastembed]'
+```python
+pip install "qdrant-client[fastembed]>=1.14.2"
```
### Step 2: Initializing the Qdrant Client
@@ -165,9 +167,9 @@ pip install 'qdrant-client[fastembed]'
After successful installation, the next step involves initializing the Qdrant Client. This can be done either in-memory or by specifying a database path:
```python
-from qdrant_client import QdrantClient
+from qdrant_client import QdrantClient, models
# Initialize the client
-client = QdrantClient(":memory:") # or QdrantClient(path="path/to/db")
+client = QdrantClient(":memory:") # or QdrantClient(path="path/to/db")
```
### Step 3: Preparing Documents, Metadata, and IDs
@@ -175,56 +177,67 @@ client = QdrantClient(":memory:") # or QdrantClient(path="path/to/db")
Once the client is initialized, prepare the text documents you wish to embed, along with any associated metadata and unique IDs:
```python
-docs = [
+docs = [
"Qdrant has Langchain integrations",
"Qdrant also has Llama Index integrations"
]
-metadata = [
+metadata = [
{"source": "Langchain-docs"},
{"source": "LlamaIndex-docs"},
]
-ids = [42, 2]
+ids = [42, 2]
```
-Note that the add method we’ll use is overloaded: If you skip the ids, we’ll generate those for you. metadata is obviously optional. So, you can simply use this too:
+### Step 4: Creating a Collection
+
+Qdrant needs to know the size and distance metric of the vectors it will store before you can add anything to it. Since FastEmbed determines the vector size for a given model, you can ask the client for it directly instead of hardcoding it:
```python
-docs = [
- "Qdrant has Langchain integrations",
- "Qdrant also has Llama Index integrations"
-]
-```
+model_name = "BAAI/bge-small-en-v1.5"
-### Step 4: Adding Documents to a Collection
-
-With your documents, metadata, and IDs ready, you can proceed to add these to a specified collection within Qdrant using the add method:
-
-```python
-client.add(
+client.create_collection(
collection_name="demo_collection",
- documents=docs,
- metadata=metadata,
- ids=ids
+ vectors_config=models.VectorParams(
+ size=client.get_embedding_size(model_name),
+ distance=models.Distance.COSINE,
+ ),
)
```
-Inside this function, Qdrant Client uses FastEmbed to make the text embedding, generate ids if they’re missing, and then add them to the index with metadata. This uses the DefaultEmbedding model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)
+### Step 5: Adding Documents to the Collection
+
+With the collection created, wrap each document in `models.Document`, telling Qdrant Client which model to embed it with, and upload the vectors, payload, and ids together:
+
+```python
+metadata_with_docs = [
+ {"document": doc, **meta} for doc, meta in zip(docs, metadata)
+]
+
+client.upload_collection(
+ collection_name="demo_collection",
+ vectors=[models.Document(text=doc, model=model_name) for doc in docs],
+ payload=metadata_with_docs,
+ ids=ids,
+)
+```
+
+Inside this call, Qdrant Client uses FastEmbed to generate the text embeddings and upload them to the collection along with the payload. This uses the model you specified: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5)

-### Step 5: Performing Queries
+### Step 6: Performing Queries
-Finally, you can perform queries on your stored documents. Qdrant offers a robust querying capability, and the query results can be easily retrieved as follows:
+Finally, you can perform queries on your stored documents. Wrap the query text in `models.Document` the same way, and use `query_points` to search:
```python
-search_result = client.query(
+search_result = client.query_points(
collection_name="demo_collection",
- query_text="This is a query document"
-)
+ query=models.Document(text="This is a query document", model=model_name),
+).points
print(search_result)
```
-Behind the scenes, we first convert the query_text to the embedding and use that to query the vector index.
+Behind the scenes, we first convert the query document to an embedding and use that to query the vector index.

@@ -244,4 +257,4 @@ So, go ahead, take it for a test drive. We're excited to hear what you think!
Lastly, If you find FastEmbed useful and want to keep up with what we're doing, giving our GitHub repo a star would mean a lot to us. Here's the link to [star the repository](https://github.com/qdrant/fastembed).
-If you ever have questions about FastEmbed, please ask them on the Qdrant Discord: [https://discord.gg/Qy6HCJK9Dc](https://discord.gg/Qy6HCJK9Dc)
+If you ever have questions about FastEmbed, please ask them on the [Qdrant Discord](https://discord.gg/qdrant).
diff --git a/qdrant-landing/content/articles/filtered-vector-search-acorn.md b/qdrant-landing/content/articles/filtered-vector-search-acorn.md
new file mode 100644
index 000000000..3b47a1a0a
--- /dev/null
+++ b/qdrant-landing/content/articles/filtered-vector-search-acorn.md
@@ -0,0 +1,158 @@
+---
+title: "Filtered Vector Search: What ACORN Fixes, and What Fixes ACORN"
+short_description: "ACORN repairs filtered HNSW search at query time, extra edges at index time. We benchmarked both in Qdrant on one million points."
+description: "Benchmark filtered vector search in Qdrant: how ACORN, filterable HNSW, and query planning trade recall for latency on one million points."
+social_preview_image: /articles_data/filtered-vector-search-acorn/preview/social_preview.jpg
+preview_dir: /articles_data/filtered-vector-search-acorn/preview
+author: Dylan Couzon & Meina Ghafouri
+date: 2026-08-07T00:00:00Z
+draft: false
+category: qdrant-internals
+weight: 5
+keywords:
+ - acorn
+ - filtered vector search
+ - filterable hnsw
+ - hnsw
+ - query planning
+---
+
+Filtered vector search breaks when metadata filters turn a healthy nearest-neighbor graph into scattered islands. HNSW's `m` parameter controls how many links each point gets. At Qdrant's default `m=16`, the one-million-point collection benchmarked below averaged about 21 links per node on layer 0. Filter out 96% of the points and fewer than one link per node survives on average, so traversal can get stranded before it reaches the true nearest matches.
+
+Qdrant repairs that damage in two places. Filterable HNSW adds extra edges at index time; ACORN steps through neighbors of neighbors at search time. Both run on the same collection. ACORN earns its cost where the extra edges don't reach: values too common to link, `AND` filters no single field's edges cover, and payload fields the build skipped silently.
+
+This benchmark runs on a single Qdrant instance and compares four of Qdrant's own search strategies over four builds.
+
+## The Two ACORNs
+
+The [ACORN paper](https://arxiv.org/abs/2403.04871) (Patel et al., SIGMOD 2024) describes two algorithms. Its headline claim of "2-1,000x higher throughput at a fixed recall" belongs to ACORN-gamma, which expands neighbor lists during index construction at 8.8x to 33.1x plain HNSW's build time in the paper's own table.
+
+ACORN-1 is lighter. It builds a standard HNSW graph, then checks neighbors of neighbors at search time where direct neighbors fail the filter. Qdrant implements ACORN-1 as a query parameter you opt into per request, with no index-time changes.
+
+## The Graph Qdrant Builds Instead
+
+[Filterable HNSW](/articles/filterable-hnsw/), which our co-founder Andrey Vasnetsov described in 2019, builds the repair into the index. When a payload field, the metadata attached to each point, is [indexed](/documentation/manage-data/indexing/#payload-index), Qdrant adds extra HNSW edges between points that share a value in that field, so a filtered query keeps a connected graph to traverse. Qdrant gives those edges to payload fields at index time, and not every field earns them.
+
+Those edges cost build time. On our one-million-point collection, the HNSW index built in 116 seconds without them and 507 to 652 seconds with them, 4.4x to 5.6x the cost. That range covers two builds at identical settings, so it is build-to-build variance. Both figures are index build time, with ingest excluded.
+
+Qdrant builds those edges per payload field, never per combination, so an `AND` filter lands on an intersection that no single field's edges cover. ACORN-1 covers that gap and pays at query time instead of build time. Qdrant's [query planner](/documentation/search/search/#query-planning) chooses automatically between ACORN, full scan, retrieval straight from the payload index, and filterable HNSW.
+
+{{< figure src="/articles_data/filtered-vector-search-acorn/two-repairs.svg" alt="Three panels of the same 12-point HNSW graph with a search path drawn in each. In the first, the path leaves a matching point and is blocked at its filtered-out neighbors. In the second, ACORN carries the path through two filtered-out neighbors to reach the other matching points. In the third, the path follows extra edges that filterable HNSW added between points sharing an indexed value." caption="The same graph, repaired two ways. ACORN steps through filtered-out neighbors at search time; filterable HNSW adds extra edges at index time that a filtered query can walk directly." width="100%" >}}
+
+## The Benchmark
+
+The benchmark runs on one million `deep-image-96` vectors, 96-dimensional image embeddings from the [ANN-benchmarks](https://github.com/erikbern/ann-benchmarks) suite. Keyword filters match from 20% of the points down to 0.012%. `Recall@10` is scored against exact brute force over 500 queries per filter, and latency is mean server-side query time.
+
+We tested four strategies:
+
+1. **Plain graph**: standard HNSW with no extra edges.
+2. **Plain graph + ACORN**: the same graph with ACORN forced on.
+3. **Filterable HNSW**: the default build with extra edges.
+4. **Planner + ACORN**: Qdrant's default query planner, free to route each query to ACORN, full scan, or the payload index.
+
+Every filter matches one keyword value on a payload field. The collection carries seven such fields, holding 5, 10, or 100 distinct values each.
+
+Most filters are independent of the vectors. The Correlated (10%) row is the easy case, where points that pass the filter also sit near each other in vector space.
+
+Every number below was measured on Qdrant v1.18.2, on one laptop-class machine, queried serially. Read the ratios, not the absolute milliseconds. The [reproduction kit](https://github.com/qdrant-labs/acorn-filterable-hnsw-benchmark) documents the hardware and the full methodology.
+
+## Single Filters: Extra Edges Win
+
+`hnsw_ef`, shortened to `ef` below, is the number of candidates the search evaluates, so raising it improves recall and slows the query. Selectivity is the fraction of points that pass the filter.
+
+This table compares the first three strategies. [`full_scan_threshold`](/documentation/manage-data/indexing/#vector-index) tells Qdrant when a filtered result set is small enough to scan directly. The value is measured in kilobytes of vector data, and Qdrant skips the HNSW graph when the matching vectors fall below it.
+We pinned it low for these three strategies so every query stayed on the graph; Planner + ACORN runs with the default threshold. Each cell shows `Recall@10` and mean server-side latency at `hnsw_ef=64`.
+
+| Filter (selectivity) | Plain graph | Plain graph + ACORN | Filterable HNSW |
+|---|---|---|---|
+| One keyword (20%) | 62.9% @ 1.6ms | 98.9% @ 4.4ms | 94.8% @ 1.2ms |
+| One keyword (10%) | 20.6% @ 1.7ms | 98.1% @ 4.3ms | 99.0% @ 1.1ms |
+| One keyword (1%) | 0.1% @ 1.6ms | 67.7% @ 4.7ms | 99.8% @ 1.0ms |
+| Correlated (10%) | 88.4% @ 1.7ms | 98.6% @ 3.5ms | 99.0% @ 1.2ms |
+
+The plain graph collapses as filters tighten, and only the correlated filter holds up. ACORN pulls recall back at 2.1x to 2.9x the plain graph's latency, then stalls on the 1% filter, the weakness the [RACORN-1 follow-up paper](https://arxiv.org/abs/2607.00768) targets. The one filter ACORN wins, at 20%, runs on a payload field that got no extra edges, and the next section explains why.
+
+{{< figure src="/articles_data/filtered-vector-search-acorn/single-filters.png" alt="Bar chart of recall at hnsw_ef=64 on four single-field filters, with each bar's mean server-side latency, comparing plain graph, plain graph with ACORN, and filterable HNSW." caption="Bars show Recall@10; the label on each bar is its mean server-side latency. Extra edges hold the top recall at about 1ms; ACORN pays 3 to 5x that." width="100%" >}}
+
+Qdrant's planner sits above all three. It estimates how many points a filter passes, then picks a path per query: the graph, ACORN on the graph, or the payload index once the estimate falls below `full_scan_threshold`. Planner + ACORN, the fourth strategy, holds 99.9% to 100% recall on all four filters, at 7.2ms to 10.9ms on the graph and 1.5ms on the 1% filter, where all 500 queries came from the payload index.
+
+## Why Some Payload Fields Get No Extra Edges
+
+Qdrant builds extra edges by walking the values of each indexed payload field. For each value it finds the points that share it and links them, so a query filtered to that value still has a graph to traverse.
+
+A value shared by more points than a size cap gets no extra edges, because the main graph should already keep that many points connected. Qdrant derives that cap per segment, the slice of a collection that has its own index. The formula is point count divided by average links per node, times four.
+Here one segment held all million points, so one million over 21 links, times four, gives 190,476 points, about 19% of the collection. Denser graphs get stricter caps: at 24 links per node, the cap falls to 16.7%.
+
+Qdrant does not report these decisions, so the reproduction kit derives them from trace-level build logs and the field sizes. The benchmark's seven payload fields landed like this:
+
+| Field | Distinct values | Points per value | Extra edges built |
+|---|---|---|---|
+| 2 fields | 5 | ~200,000 | No, all 5 values over the cap |
+| 2 fields | 10 | ~100,000 | Yes, 10 of 10 values |
+| Correlated field | 10 | ~100,000 | Yes, 10 of 10 values |
+| 2 fields | 100 | ~10,000 | Yes, 100 of 100 values |
+
+The 5-value fields sit 5% over the cap, so every one of their values was skipped. That skip is why ACORN beats filterable HNSW on the 20% filter, and on the 4% intersection in the next section. Everywhere else the gap stays within build-to-build variance.
+
+Skipping is deliberate: extra edges cost build time and memory, which is why the cap exists. A value under the cap can still be skipped when it sits below the `full_scan_threshold` floor or fails a sampled check of how well its points already connect, so the value count alone does not decide the outcome.
+
+## Double Filters: The Intersection Gap
+
+The same benchmark at `hnsw_ef=64`, now with an `AND` filter over two keyword fields.
+
+| Filter (selectivity) | Plain graph + ACORN | Filterable HNSW | Planner + ACORN |
+|---|---|---|---|
+| Two keywords (4%) | 95.2% @ 7.7ms | 63.7% @ 1.2ms | 99.9% @ 13.9ms |
+| Two keywords (1%) | 72.7% @ 6.8ms | 70.8% @ 1.5ms | 100% @ 3.7ms |
+| Two keywords (0.012%) | 0.6% @ 2.6ms | 1.8% @ 2.6ms | 100% @ 1.3ms |
+
+A two-keyword intersection has no extra edges of its own, even when both its payload fields do. Neither repair closes the gap at this `ef`. On the 1% row, ACORN's recall spans 70.7% to 74.1% across rebuilds of the same graph, wider than its lead in the table.
+The plain graph, dropped from this table, scored 0.1% and 0.0% on the first two rows, and `ef=512` changes nothing once traversal has exhausted its disconnected island.
+
+Raising `ef` breaks the tie on the 1% intersection. At `ef=512`, filterable HNSW reaches 91.2% recall at 4.9ms while ACORN needs 20.1ms to reach 90.3%. Repairing the graph at search time costs four times the latency for slightly less recall here. The 4% intersection is the exception, where both fields exceeded the cap and ACORN leads 99.6% to 92.5%.
+
+{{< figure src="/articles_data/filtered-vector-search-acorn/ef-sweep.png" alt="Recall versus server-side latency for four filtered-search strategies as hnsw_ef sweeps from 64 to 512." caption="Recall vs server-side latency on the 1% double filter alone, hnsw_ef swept from 64 to 512." width="100%" >}}
+
+At 0.012%, roughly 120 points match in a million, and the graph stops being the right tool. Planner + ACORN wins that row by reading the payload index instead. The choice happens per query: on the 1% intersection it sent 29 of the 500 queries to the graph and 471 to the payload index, and at 4% it stayed on the graph throughout.
+
+## ACORN on a Normal Collection
+
+The earlier tables pinned `full_scan_threshold` low to hold the three fixed strategies on the graph. Nobody runs a collection that way. This is the default configuration: extra edges, the default threshold, and the planner free to choose the graph or the payload index in both columns. ACORN is off by default, so the left column is what a collection with payload indexes returns today.
+
+| Filter (selectivity) | Planner, ACORN off | Planner + ACORN |
+|---|---|---|
+| One keyword (20%) | 90.8% @ 1.1ms | 100% @ 5.7ms |
+| One keyword (10%) | 98.6% @ 0.9ms | 99.9% @ 4.4ms |
+| One keyword (1%) | 100% @ 1.7ms | 100% @ 1.6ms |
+| Correlated (10%) | 98.6% @ 1.0ms | 100% @ 4.2ms |
+| Two keywords (4%) | 39.7% @ 1.1ms | 100% @ 7.3ms |
+| Two keywords (1%) | 97.2% @ 2.1ms | 100% @ 2.5ms |
+| Two keywords (0.012%) | 100% @ 1.4ms | 100% @ 1.2ms |
+
+Most filters need no help: the planner sends highly selective filters straight to the payload index, and extra edges carry the broad ones on the graph. ACORN earns its place on the two middle cases, kept on the graph with no extra edges on their payload fields. It adds 9 percentage points on the 20% filter and 60 percentage points on the 4% intersection.
+
+When the planner stays on the graph, ACORN is expensive: 5.4x latency on the 20% filter and 6.7x on the 4% intersection. When it routes most queries to the payload index instead, ACORN is nearly free: the 1% intersection gains 2.8 percentage points at 1.2x the latency because 471 of its 500 queries never touch the graph.
+
+Extra edges also make ACORN stronger. On the 4% intersection it reached 95.2% on the plain graph and 99.9% with the edges in place. ACORN traverses the graph it gets, so the two repairs stack.
+
+## What to Measure on Your Own Collection
+
+Measure recall for each filter shape you serve. Start with the ones most likely to break: values covering roughly a fifth of the collection or more, and `AND` combinations of them. [Facet counts](/documentation/manage-data/payload/#facet-counts) show which values are that broad.
+On the default configuration here, one filter returned 39.7% with ACORN off while every other filter stayed above 90%, and a single aggregate number would have hidden it. If those filters come back clean, test narrower values next.
+
+Create a payload index on every field you filter on, and leave ACORN off to start, since that is Qdrant's default. Then sample a few hundred real queries per filter shape, 500 if you want to match this benchmark. Get exact results with `exact: true`, and score both recall and latency with ACORN off and then on.
+
+Compare the recall gain with the latency cost. A recovery like that 39.7% filter is what ACORN is for, while a point or two is worth taking only when the latency multiple is small. Set [`acorn.enable`](/documentation/search/search/#acorn-search-algorithm) on the query paths whose filters earned it.
+Turning it on never lowered recall in any of our runs, so if you are unsure, the cost of leaving it on is latency. Qdrant applies it only below `max_selectivity`, 0.4 by default, so a filter matching half your collection will not change either way.
+
+Extra edges fix the graph before a query ever arrives; ACORN fixes the gaps that remain.
+
+## Further Reading
+
+- [ACORN (Patel et al., SIGMOD 2024)](https://arxiv.org/abs/2403.04871): the paper behind ACORN-gamma and ACORN-1.
+- [RACORN-1](https://arxiv.org/abs/2607.00768): a follow-up targeting ACORN-1's recall collapse at low selectivity.
+- [PostgreSQL ACORN study](https://arxiv.org/abs/2603.23710): measures the filter-check cost of the search-time repair.
+- [Filterable HNSW](/articles/filterable-hnsw/): the 2019 article behind Qdrant's extra edges.
+- [Reproduction kit](https://github.com/qdrant-labs/acorn-filterable-hnsw-benchmark): scripts, pinned image, and ground truth to re-run these tables against your Qdrant version.
+
+To discuss your filtered-search setup, [get in touch](/contact-us/).
diff --git a/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md b/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md
index 7ddad9de0..32378d17f 100644
--- a/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md
+++ b/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md
@@ -28,13 +28,13 @@ Selecting the best embedding model is a multi-objective optimization problem and
and there probably never will be. In this article, we will try to provide some guidance on how to approach this problem
in a practical way, and how to move from model selection to running it in production.
-## Evaluation: the holy grail of vector search
+## Evaluation: The Holy Grail of Vector Search
You can't improve what you don't measure. It's cliché, but it's true also for retrieval. Search quality might and should
be measured not only in a running system, but also before you make the most important decision - which embedding model
to use.
-### Know the language your model speaks
+### Know the Language Your Model Speaks
Embedding models are trained with specific languages in mind. When evaluating one, consider whether it supports all the
languages you have or predict to have in your data. If your data is not homogeneous, you might require a multilingual
@@ -79,7 +79,7 @@ similarity between the original and modified text.
If the created representations are really far from each other in the vector space, it may indicate that some
non-supported characters are replaced with `UNK` tokens and thus the model can't properly embed the input data.
-### Checklist of things to consider
+### Checklist of Things to Consider
Nevertheless, the evaluation does not focus on the input tokens only. First and foremost, we should measure how well
a particular model can handle the task we want to use it for. Vector embeddings are multipurpose tools, and some models
@@ -100,7 +100,7 @@ The list is not exhaustive, as there might be plenty of other things to consider
That's why you need to precisely define the task you really want to solve, get your hands dirty with the data the system
is supposed to process and build a ground truth dataset for it, so you can make an informed decision.
-### Building the ground truth dataset
+### Building the Ground Truth Dataset
The way your dataset will look like depends on the task you want to evaluate. If we speak about semantic similarity,
then you will need pairs of texts with a score indicating how similar they are.
@@ -168,11 +168,11 @@ help you with that. [Running the evaluation process](/rag/rag-evaluation-guide/)
a sense of how they perform on your data. You can test even proprietary models that way. However, it's not the only
thing you should consider when choosing the best model.
-Please do not be afraid of building your evaluation dataset. It’s not as complicated as it might seem, and it's a
-critical step! You don’t need millions of samples to get a good idea of how the model performs. A few hundred
+Please do not be afraid of building your evaluation dataset. It's not as complicated as it might seem, and it's a
+critical step! You don't need millions of samples to get a good idea of how the model performs. A few hundred
well-curated examples might be a good starting point. Even dozens are better than nothing!
-## Compute resource constraints
+## Compute Resource Constraints
Even if you found the best performing embedding model for your domain, that doesn't mean you can use it. Software projects
do not live in isolation, and you have to consider the bigger picture. For example, you might have budget constraints
@@ -182,7 +182,7 @@ slower and consumes 10 times more resources, is it really worth it?
Eventually, enjoying the journey is more important than reaching the destination in some cases, but that doesn't hold
true for search. The simpler and faster the means that took you there, the better.
-## Throughput, latency and cost
+## Throughput, Latency and Cost
When selecting an embedding model for production, you need to consider three critical operational factors:
@@ -201,7 +201,7 @@ processing large volumes of articles in real-time, while a website search might
results. Similarly, a chatbot using a Large Language Model to generate a response might prioritize cost-effectiveness,
as LLMs are often slower and retrieval isn't the most time-consuming part of the process.
-## Balancing all aspects
+## Balancing All Aspects
After all these considerations, you should have a table that summarizes each of the models you evaluated under all the
different conditions. Now things are getting hard and answers are not obvious anymore.
@@ -227,10 +227,14 @@ choice of the embedding model. Qdrant's architecture makes it relatively easy to
Named vectors help to create a system with multiple models and switch between them based on the query, or build a
[hybrid search](/articles/hybrid-search/) that takes advantage of different models or more complex search pipelines.
+Choosing the right embedding model is one of the most important design decisions in a vector search system, but it is just one of several levers.
+Memory usage can often be reduced with techniques such as quantization or Matryoshka embeddings, while retrieval quality may benefit more from hybrid search or reranking than from switching to a larger embedding model.
+The key takeaway is that while the embedding model matters a great deal, cost, retrieval quality, latency, and throughput are properties of the retrieval pipeline and system as a whole.
+
An important decision to make is also where to host the embedding model. Maybe you prefer not to deal with the
infrastructure management and send the data you process in its original form? Qdrant now has something for you!
-## Locally sourced embeddings
+## Locally Sourced Embeddings
Wouldn't it be great to run your selected embedding model as close to your search engine as possible? Network latency
might be one of the biggest enemies, and transferring millions of vectors over the network may take longer if done from
diff --git a/qdrant-landing/content/articles/how-to-tune-hybrid-search.md b/qdrant-landing/content/articles/how-to-tune-hybrid-search.md
new file mode 100644
index 000000000..d90f03982
--- /dev/null
+++ b/qdrant-landing/content/articles/how-to-tune-hybrid-search.md
@@ -0,0 +1,179 @@
+---
+title: "How to Tune Hybrid Search in Qdrant"
+short_description: "Tune hybrid search with RRF or DBSF, choose k from relevance labels, and learn why weights are pairs instead of ratios."
+description: "Tune hybrid search fusion in Qdrant: choose between RRF and DBSF, set the constant k from your relevance labels, and get weights right."
+preview_dir: /articles_data/how-to-tune-hybrid-search/preview
+social_preview_image: /articles_data/how-to-tune-hybrid-search/preview/social_preview.jpg
+weight: -211
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-22T00:00:00+03:00
+draft: false
+keywords:
+ - hybrid search tuning
+ - reciprocal rank fusion
+ - RRF k parameter
+ - fusion weights
+ - DBSF
+category: search-quality
+---
+
+Before you tune fusion, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline.
+
+Hybrid search retrieves dense and sparse candidate lists, then fuses them into one ranking. The dense prefetch finds similar meaning; the sparse prefetch finds matching keywords. Fusion reorders the candidates the prefetches return, so a document missing from both lists cannot appear in the result.
+
+## Confirm Fusion Beats Either Prefetch
+
+Before tuning, compare dense retrieval, sparse retrieval, and default [Reciprocal Rank Fusion](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf) (RRF) at `k=2` and equal weights. Score all three with `nDCG@10`, which grades the top 10 results and gives more credit to relevant documents near the top.
+
+Qdrant defaults to `k=2`. The original RRF paper uses 60, which maps to `k=61` in Qdrant's formula. That gap is what most of this article is about.
+
+
+
+`Over the Better One` is default RRF's `nDCG@10` minus the better individual prefetch. `Second Prefetch Cost` is the median latency the second prefetch adds over the dense prefetch alone.
+
+| Dataset | Dense Alone | Sparse Alone | Both, RRF (`k=2`) | Over the Better One | Second Prefetch Cost |
+|---|---|---|---|---|---|
+| SciFact | 0.6239 | 0.6886 | 0.7175 | +0.0289 | +0.73 ms |
+| ArguAna | 0.4905 | 0.4224 | 0.5216 | +0.0311 | +1.47 ms |
+| WANDS | 0.6921 | 0.7098 | 0.7254 | +0.0156 | +0.60 ms |
+| CodeSearchNet | 0.6299 | 0.5126 | 0.6555 | +0.0256 | +0.68 ms |
+| DBPedia-entity | 0.4677 | 0.3857 | 0.4638 | -0.0039 | +0.64 ms |
+
+Fusion outscored both prefetches in four datasets, and each gain's 95% interval excludes zero. DBPedia-entity is the exception: fusion trails dense retrieval by 0.0039, and its interval crosses zero.
+
+The second prefetch also needs a second index and a second vector per point. Keep it when it improves relevance on your own labels.
+
+## RRF and DBSF Use Different Signals
+
+[Reciprocal Rank Fusion](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf) (RRF) uses only a candidate's position in each prefetch. A document at rank 1 scores the same whether it beat rank 2 by a wide margin or a narrow one. [Distribution-based score fusion](/documentation/search/hybrid-queries/#distribution-based-score-fusion-dbsf) (DBSF) puts both lists on one scale for each query, using each list's average score and how spread out its scores are. Adding the two rescaled scores carries the size of a lead into the fused ranking, and a document only one prefetch retrieved keeps that single rescaled score.
+
+
+
+_RRF reads each document's slot, so A's dense lead flattens to one step and B, ranked near the top by both prefetches, wins. DBSF keeps the spacing on a shared axis, so A's lead survives the sum and A wins._
+
+RRF ignores score scale, so a cosine similarity and a BM25 score combine without either dominating. DBSF assumes the size of a score gap means something, so one outlying score can move the result. Which one wins depends on your data, so run both against your labels.
+
+## Compare RRF and DBSF on Your Labels
+
+Use your [labeled query set](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to compare RRF and DBSF over the same prefetches. Run RRF at `k=2` and equal weights, then run DBSF.
+
+Both queries read the same two candidate lists, so connect once and build the prefetches once. The prefetches must use the models the collection was indexed with.
+
+```python
+from qdrant_client import QdrantClient, models
+from your_embedding_setup import dense_query, sparse_query
+
+client = QdrantClient(
+ url="https://YOUR-CLUSTER.cloud.qdrant.io",
+ api_key="",
+)
+
+dense_prefetch = models.Prefetch(query=dense_query, using="dense", limit=200)
+sparse_prefetch = models.Prefetch(query=sparse_query, using="bm25", limit=200)
+prefetches = [dense_prefetch, sparse_prefetch]
+```
+
+`RrfQuery` carries both RRF settings, `k` and the weight pair, shown here at their defaults. It requires Qdrant v1.17 or later and a compatible `qdrant-client` release.
+
+```python
+rrf_response = client.query_points(
+ collection_name="products",
+ prefetch=prefetches,
+ query=models.RrfQuery(rrf=models.Rrf(k=2, weights=[1.0, 1.0])),
+ limit=10,
+)
+```
+
+The DBSF query differs only in the fusion step.
+
+```python
+dbsf_response = client.query_points(
+ collection_name="products",
+ prefetch=prefetches,
+ query=models.FusionQuery(fusion=models.Fusion.DBSF),
+ limit=10,
+)
+```
+
+
+
+On three of these five datasets, DBSF scored higher than default RRF by a margin whose 95% interval excludes zero. SciFact's 0.0148 gain and ArguAna's 0.0045 loss both cross zero, so those two datasets are inconclusive.
+
+| Dataset | DBSF | Over Default RRF |
+|---|---|---|
+| ArguAna | 0.5171 | -0.0045 |
+| CodeSearchNet | 0.6716 | +0.0161 |
+| SciFact | 0.7323 | +0.0148 |
+| DBPedia-entity | 0.4822 | +0.0184 |
+| WANDS | 0.7637 | +0.0383 |
+
+DBSF takes no parameters: `k` and the weight pair are RRF settings, and the public API accepts them only on an `RrfQuery`. So if DBSF wins on your labels, skip the next two sections and go to the held-out check.
+
+## Use Labels to Choose a `k` Range
+
+Qdrant scores a document at position `pos` in one prefetch as `1 / ((pos + 1) / weight + k - 1)`, then sums across prefetches. With equal weights that reduces to `1 / (pos + k)`, and `k` alone decides how steeply the head of a list outranks its tail.
+
+
+
+_At Qdrant's default of k=2, rank 1 carries 5.50 times the score weight of rank 10. At k=61, it carries 1.15 times the weight, so a candidate's presence in a prefetch matters almost as much as its position._
+
+Rank 1 outweighs rank 10 by 2.80 times at `k=5` and 1.45 times at `k=20`, so most of the movement sits below `k=20`. A sweep in even steps of five would spend most of its runs past the point where the curve stops moving.
+
+Sweep `k` over 1, 2, 5, 20, and 61, changing only `k` in `models.Rrf` and keeping equal weights. Lower values favor a document one prefetch ranks highly, and higher values give more credit to documents both prefetches retrieve.
+
+The table gives `nDCG@10` at equal weights across five values of `k`, with `k=2` as default RRF. A star marks the best `k` in each row.
+
+| Dataset | Queries | Relevant per Query | k=1 | k=2 | k=5 | k=20 | k=61 |
+|---|---|---|---|---|---|---|---|
+| ArguAna | 1,401 | 1.0 | 0.5171 | 0.5216 | 0.5304* | 0.5269 | 0.5207 |
+| CodeSearchNet | 1,000 | 1.0 | 0.6501 | 0.6555 | 0.6580* | 0.6511 | 0.6258 |
+| SciFact | 300 | 1.1 | 0.7117 | 0.7175* | 0.7154 | 0.7122 | 0.7067 |
+| DBPedia-entity | 400 | 38.2 | 0.4625 | 0.4638 | 0.4641 | 0.4682* | 0.4606 |
+| WANDS | 480 | 358.9 | 0.7232 | 0.7254 | 0.7336 | 0.7571 | 0.7614* |
+
+On WANDS, `k=2` and `k=61` chose a different top result for 42% of queries, while `nDCG@10` rose by 0.0360. A small aggregate gain can still change what a user sees first.
+
+These five datasets suggest a direction: with about one relevant document per query, the best `k` was 2 or 5; with tens or hundreds, it was 20 or 61. Count relevant documents per query in your labeled query set, then try that part of the range first.
+
+If you are porting an RRF configuration from another system, remember that Qdrant uses zero-based positions. To reproduce [the `1 / (rank + 60)` convention from Cormack et al.](https://dl.acm.org/doi/10.1145/1571941.1572114) with one-based ranks, use `k=61`.
+
+
+
+## Tune Weights Last
+
+A weight pair gives one multiplier to each prefetch, in the order the prefetches appear in the query. The pair is absolute, so `(1, 2)` and `(2, 4)` are two different settings: the formula divides the position by the weight, so scaling both weights changes every score. On WANDS at `k=5`, `(1, 2)` scores 0.7390 and `(2, 4)` scores 0.7508.
+
+Settle `k` first, since a pair is only valid for the `k` you tested it with. On WANDS, `(2, 4)` beats equal weights at `k=5`. At `k=61`, that dataset's best value, equal weights win: 0.7614 against 0.7567.
+
+Then sweep a few pairs and let your labels pick the winner. A prefetch's own score does not say which way to lean. Weights act on positions inside each list, so the pair is decided by which prefetch ranks relevant documents highly on the queries the other one misses.
+
+On DBPedia-entity, dense retrieval scores 0.4677 against sparse retrieval's 0.3857, yet the winning pair `(1, 3)` gives sparse three times the dense weight and gains 0.0060. CodeSearchNet leans the other way and gains 0.0096 at `(2, 1)`. Both intervals exclude zero.
+
+Equal weights are a real outcome. Six pairs ran at each dataset's best `k`, and `(1, 1)` won outright on two of the five. ArguAna's best pair gained 0.0029, with an interval that crosses zero.
+
+A weight of 0.0 keeps every document from that prefetch and scores each one 0.0. The documents stay at the bottom of the fused list instead of disappearing.
+
+## Confirm the Selected Configuration on Held-Out Queries
+
+A configuration can score best on the queries used to select it and still fail on held-out queries. Run both checks from [the pre-tuning article](/articles/before-tuning-a-qdrant-collection/): a bootstrap interval on per-query gain, and a split between selection and held-out queries. Ship a configuration when its interval excludes zero and its selected gain holds on the held-out half.
+
+On SciFact's 300 queries, nothing we tried had a 95% interval that excluded zero, including DBSF's 0.0148 gain. Across 200 random splits, a selected fusion configuration kept 67% to 95% of its gain on held-out queries. Keeping the default is a real answer, and it was the right one on one of our five datasets.
+
+## Tune in This Order
+
+Each step is cheap enough to run in a single session.
+
+1. Confirm fusion beats either prefetch alone.
+2. Pick RRF or DBSF on your labels.
+3. Set `k` from the number of relevant documents per query.
+4. Sweep a few weight pairs at that `k`.
+5. Validate the winner on held-out queries before shipping.
+
+Next, if a downstream model could improve the ranking of your retrieved candidates, [test whether a reranker is worth its cost](/articles/when-a-reranker-is-worth-it/).
diff --git a/qdrant-landing/content/articles/hybrid-search.md b/qdrant-landing/content/articles/hybrid-search.md
index ee411dd2e..6a4414402 100644
--- a/qdrant-landing/content/articles/hybrid-search.md
+++ b/qdrant-landing/content/articles/hybrid-search.md
@@ -1,409 +1,156 @@
---
-title: "Hybrid Search with Qdrant's Query API"
-short_description: "Merging different search methods to improve the search quality was never easier"
-description: "Our new Query API allows you to build a hybrid search system that uses different search methods to improve search quality & experience. Learn more here."
+title: "Hybrid Search in Qdrant"
+short_description: "Run dense and sparse retrieval together: the queries each one gets wrong, what the second index costs, and how to tell if it helped."
+description: "Decide whether to add hybrid search in Qdrant: the queries dense and sparse retrieval each get wrong, and how to measure the gain."
preview_dir: /articles_data/hybrid-search/preview
social_preview_image: /articles_data/hybrid-search/preview/social_preview.jpg
-weight: 80
-author: Kacper Łukawski
-author_link: https://kacperlukawski.com
-date: 2024-07-25T00:00:00.000Z
-category: mastering-search
+weight: -215
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-24T09:00:00+03:00
+draft: false
+keywords:
+ - hybrid search
+ - sparse vectors
+ - BM25
+ - reciprocal rank fusion
+ - search relevance
+category: search-quality
---
-It's been over a year since we published the original article on how to build a hybrid
-search system with Qdrant. The idea was straightforward: combine the results from different search methods to improve
-retrieval quality. Back in 2023, you still needed to use an additional service to bring lexical search
-capabilities and combine all the intermediate results. Things have changed since then. Once we introduced support for
-sparse vectors, [the additional search service became obsolete](/articles/sparse-vectors/), but you were still
-required to combine the results from different methods on your end.
+A search result can look plausible and still be wrong. Dense retrieval can return a document on the right topic but miss an exact identifier copied into the query. Sparse retrieval can miss a relevant document when the query describes it with terms the corpus doesn't use. Either way, your logs record a successful query.
-**Qdrant 1.10 introduces a new Query API that lets you build a search system by combining different search methods
-to improve retrieval quality**. Everything is now done on the server side, and you can focus on building the best search
-experience for your users. In this article, we will show you how to utilize the new [Query
-API](/documentation/search/search/#query-api) to build a hybrid search system.
+Hybrid search runs dense and sparse retrieval over the same query, then merges their result lists. Dense retrieval adds semantic similarity, so paraphrases can rank together. Sparse retrieval adds weighted term matching for exact words and identifiers.
-## Introducing the new Query API
+Compared with either retriever alone, hybrid search adds storage, indexing, and query work. Measure whether the gain is worth the cost instead of guessing.
-At Qdrant, we believe that vector search capabilities go well beyond a simple search for nearest neighbors.
-That's why we provided separate methods for different search use cases, such as `search`, `recommend`, or `discover`.
-With the latest release, we are happy to introduce the new Query API, which combines all of these methods into a single
-endpoint and also supports creating nested multistage queries that can be used to build complex search pipelines.
+## Dense and Sparse Retrieval Miss Different Things
-If you are an existing Qdrant user, you probably have a running search mechanism that you want to improve, whether sparse
-or dense. Doing any changes should be preceded by a proper evaluation of its effectiveness.
+Dense retrieval embeds the query and each document, then ranks the documents by vector similarity. The model can place paraphrases near each other, but exact strings may lose influence among documents with similar meanings.
-## How effective is your search system?
+Sparse retrieval represents text as weighted terms and scores the overlap between the query and document. BM25 sets those weights from term frequency, inverse document frequency, and document length. It requires no model inference.
-None of the experiments makes sense if you don't measure the quality. How else would you compare which method works
-better for your use case? The most common way of doing that is by using the standard metrics, such as `precision@k`,
-`MRR`, or `NDCG`. There are existing libraries, such as [ranx](https://amenra.github.io/ranx/), that can help you with
-that. We need to have the ground truth dataset to calculate any of these, but curating it is a separate task.
+The product-search examples make that difference concrete. For each query, one retriever ranks a relevant product first, while the other ranks an irrelevant product first.
-```python
-from ranx import Qrels, Run, evaluate
+| Query | Dense Retrieval | Sparse Retrieval |
+|---|---|---|
+| french molding | french curves 6'' h x 6'' w x 1'' d rosette applique (Relevant) | french bread mold toast tray non-stick tray baking tray (Irrelevant) |
+| wayfair comforters | wayfair basics comforter set (Relevant) | wayfair basics peva shower curtain liner (Irrelevant) |
+| bathroom vanity knobs | carran 30'' single bathroom vanity set (Irrelevant) | damask mushroom knob (Relevant) |
+| farmhouse cabinet | rustic storage cabinet (Irrelevant) | farmhouse 2 door accent cabinet (Relevant) |
-# Qrels, or query relevance judgments, keep the ground truth data
-qrels_dict = { "q_1": { "d_12": 5, "d_25": 3 },
- "q_2": { "d_11": 6, "d_22": 1 } }
+For "french molding," sparse retrieval follows the terms "french" and "mold" to the wrong product. For "bathroom vanity knobs," dense retrieval finds the right category, while sparse retrieval follows "knobs" to the relevant product.
-# Runs are built from the search results
-run_dict = { "q_1": { "d_12": 0.9, "d_23": 0.8, "d_25": 0.7,
- "d_36": 0.6, "d_32": 0.5, "d_35": 0.4 },
- "q_2": { "d_12": 0.9, "d_11": 0.8, "d_25": 0.7,
- "d_36": 0.6, "d_22": 0.5, "d_35": 0.4 } }
+Learned sparse models change what the sparse side matches. [miniCOIL](/documentation/fastembed/fastembed-minicoil/) keeps BM25's term matching but reweights each term by context, so "bat" in a sports listing and "bat" in a wildlife guide no longer share one weight. [SPLADE](/documentation/fastembed/fastembed-splade/) adds related terms that the text never used. This recovers synonyms and moves sparse retrieval closer to what the dense retriever already covers. Start with BM25, which needs no model at query time, then measure a learned model against it before adopting one.
-# We need to create both objects, and then we can evaluate the run against the qrels
-qrels = Qrels(qrels_dict)
-run = Run(run_dict)
+## Fusion Merges Two Rankings Into One
-# Calculating the NDCG@5 metric is as simple as that
-evaluate(qrels, run, "ndcg@5")
-```
+In Qdrant, a prefetch runs a search and passes its candidates to the main query. Hybrid search uses one prefetch for dense retrieval and another for sparse retrieval. Fusion combines their candidate lists into one ordering.
-## Available embedding options with Query API
+Dense similarity and BM25 scores use different scales. Dense similarity is bounded, while BM25's magnitude depends on how many query terms match and how rare they are in the corpus. A fixed weight on the raw scores may balance one query but let BM25 dominate another. No single raw-score weight preserves the same balance across both.
-Support for multiple vectors per point is nothing new in Qdrant, but introducing the Query API makes it even
-more powerful. The 1.10 release supports the multivectors, allowing you to treat embedding lists
-as a single entity. There are many possible ways of utilizing this feature, and the most prominent one is the support
-for late interaction models, such as [ColBERT](https://qdrant.tech/documentation/fastembed/fastembed-colbert/). Instead of having a single embedding for each document or query, this
-family of models creates a separate one for each token of text. In the search process, the final score is calculated
-based on the interaction between the tokens of the query and the document. Contrary to cross-encoders, document
-embedding might be precomputed and stored in the database, which makes the search process much faster. If you are
-curious about the details, please check out [the article about ColBERT, written by our friends from Jina
-AI](https://jina.ai/news/what-is-colbert-and-late-interaction-and-why-they-matter-in-search/).
+
-
+_The dense scale stays similar, but the BM25 scale shifts across queries._
-Besides multivectors, you can use regular dense and sparse vectors, and experiment with smaller data types to reduce
-memory use. Named vectors can help you store different dimensionalities of the embeddings, which is useful if you
-use multiple models to represent your data, or want to utilize the Matryoshka embeddings.
+RRF avoids the scale mismatch by discarding score magnitude. DBSF normalizes each score distribution per query.
-
+Reciprocal Rank Fusion, or RRF, reads only where each document landed in each list. That lets it combine a cosine similarity of 0.7 with a BM25 score of 12.4 without comparing the values directly. [Cormack, Clarke, and Buettcher](https://dl.acm.org/doi/10.1145/1571941.1572114) introduced the method in 2009, and it remains a standard way to combine ranked lists.
-There is no single way of building a hybrid search. The process of designing it is an exploratory exercise, where you
-need to test various setups and measure their effectiveness. Building a proper search experience is a
-complex task, and it's better to keep it data-driven, not just rely on the intuition.
+Distribution-Based Score Fusion, or DBSF, rescales each list using its average score and score spread, then adds the rescaled scores. This preserves the size of score gaps, so a strong lead from one retriever can affect the final ranking.
-## Fusion vs reranking
+Neither method wins universally. Start with RRF, then compare DBSF against the same labeled queries.
-We can, distinguish two main approaches to building a hybrid search system: fusion and reranking. The former is about
-combining the results from different search methods, based solely on the scores returned by each method. That usually
-involves some normalization, as the scores returned by different methods might be in different ranges. After that, there
-is a formula that takes the relevancy measures and calculates the final score that we use later on to reorder the
-documents. Qdrant has built-in support for the Reciprocal Rank Fusion method, which is the de facto standard in the
-field.
+Formula Queries serve a different purpose: they rescore retrieved candidates with an expression over their retrieval scores and payload values. For example, a formula can boost recent or in-stock items. It does not make unnormalized dense and BM25 scores directly comparable. [Custom scoring](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) covers the expression syntax.
-
+Fusion only reorders. It works on the union of what the two prefetches returned, so a document neither one found cannot appear anywhere in the results.
-Reranking, on the other hand, is about taking the results from different search methods and reordering them based on
-some additional processing using the content of the documents, not just the scores. This processing may rely on an
-additional neural model, such as a cross-encoder which would be inefficient enough to be used on the whole dataset.
-These methods are practically applicable only when used on a smaller subset of candidates returned by the faster search
-methods. Late interaction models, such as ColBERT, are way more efficient in this case, as they can be used to rerank
-the candidates without the need to access all the documents in the collection.
+If a relevant document falls below a prefetch cutoff, increasing one or both prefetch limits can expose it to fusion. A larger limit adds retrieval work, and it does not help if the retrievers still miss the document at greater depth. [Candidate depth](/articles/candidate-depth/) explains how to test the limits, and the [hybrid query documentation](/documentation/search/hybrid-queries/) covers how prefetches feed fusion.
-
+
-### Why not a linear combination?
+_The pale documents were never retrieved. If the right answer is one of them, no fusion method reaches it._
-It's often proposed to use full-text and vector search scores to form a linear combination formula to rerank
-the results. So it goes like this:
+## What a Second Retriever Costs
-```final_score = 0.7 * vector_score + 0.3 * full_text_score```
+The setup that follows starts with a dense-only collection and adds BM25 as the second retriever. That means adding a sparse vector per point, a second index, and another search on every query. On one container serving one request at a time, the extra search raised median query latency by 0.60 to 1.47 ms. Measure the cost under your own concurrency and shard layout.
-However, we didn't even consider such a setup. Why? Those scores don't make the problem linearly separable. We used
-the BM25 score along with cosine vector similarity to use both of them as points coordinates in 2-dimensional space. The
-chart shows how those points are distributed:
+Adding a sparse vector to an existing dense-only collection requires a new collection and a full reindex because the vector configuration is fixed at collection creation.
-
-
-*A distribution of both Qdrant and BM25 scores mapped into 2D space. It clearly shows relevant and non-relevant
-objects are not linearly separable in that space, so using a linear combination of both scores won't give us
-a proper hybrid search.*
-
-Both relevant and non-relevant items are mixed. **None of the linear formulas would be able to distinguish
-between them.** Thus, that's not the way to solve it.
-
-## Building a hybrid search system in Qdrant
-
-Ultimately, **any search mechanism might also be a reranking mechanism**. You can prefetch results with sparse vectors
-and then rerank them with the dense ones, or the other way around. Or, if you have Matryoshka embeddings, you can start
-with oversampling the candidates with the dense vectors of the lowest dimensionality and then gradually reduce the
-number of candidates by reranking them with the higher-dimensional embeddings. Nothing stops you from
-combining both fusion and reranking.
-
-Let's go a step further and build a hybrid search mechanism that combines the results from the
-Matryoshka embeddings, dense vectors, and sparse vectors and then reranks them with the late interaction model. In the
-meantime, we will introduce additional reranking and fusion steps.
-
-
-
-Our search pipeline consists of two branches, each of them responsible for retrieving a subset of documents that
-we eventually want to rerank with the late interaction model. Let's connect to Qdrant first and then build the search
-pipeline.
+The new collection declares both vector types. The sparse vector needs the IDF modifier, which gives rare terms more weight than common ones. Without it, a common word can count as much as a part number.
```python
from qdrant_client import QdrantClient, models
-client = QdrantClient("http://localhost:6333")
-```
+client = QdrantClient(
+ url="https://YOUR-CLUSTER.cloud.qdrant.io",
+ api_key="",
+)
-All the steps utilizing Matryoshka embeddings might be specified in the Query API as a nested structure:
-
-```python
-# The first branch of our search pipeline retrieves 25 documents
-# using the Matryoshka embeddings with multistep retrieval.
-matryoshka_prefetch = models.Prefetch(
- prefetch=[
- models.Prefetch(
- prefetch=[
- # The first prefetch operation retrieves 100 documents
- # using the Matryoshka embeddings with the lowest
- # dimensionality of 64.
- models.Prefetch(
- query=[0.456, -0.789, ..., 0.239],
- using="matryoshka-64dim",
- limit=100,
- ),
- ],
- # Then, the retrieved documents are re-ranked using the
- # Matryoshka embeddings with the dimensionality of 128.
- query=[0.456, -0.789, ..., -0.789],
- using="matryoshka-128dim",
- limit=50,
- )
- ],
- # Finally, the results are re-ranked using the Matryoshka
- # embeddings with the dimensionality of 256.
- query=[0.456, -0.789, ..., 0.123],
- using="matryoshka-256dim",
- limit=25,
+client.create_collection(
+ collection_name="products",
+ vectors_config={
+ # size matches your dense model's output dimensions.
+ "dense": models.VectorParams(size=384, distance=models.Distance.COSINE)
+ },
+ sparse_vectors_config={
+ "bm25": models.SparseVectorParams(modifier=models.Modifier.IDF)
+ },
)
```
-Similarly, we can build the second branch of our search pipeline, which retrieves the documents using the dense and
-sparse vectors and performs the fusion of them using the Reciprocal Rank Fusion method:
+The [hybrid search documentation](/documentation/search/text-search/hybrid-search/) covers indexing text and producing BM25 sparse vectors in every supported language.
```python
-# The second branch of our search pipeline also retrieves 25 documents,
-# but uses the dense and sparse vectors, with their results combined
-# using the Reciprocal Rank Fusion.
-sparse_dense_rrf_prefetch = models.Prefetch(
+from your_embedding_models import dense_embed, sparse_embed
+
+query_text = "Samsung Galaxy S24 Ultra 512GB"
+
+results = client.query_points(
+ collection_name="products",
prefetch=[
models.Prefetch(
- prefetch=[
- # The first prefetch operation retrieves 100 documents
- # using dense vectors using integer data type. Retrieval
- # is faster, but quality is lower.
- models.Prefetch(
- query=[7, 63, ..., 92],
- using="dense-uint8",
- limit=100,
- )
- ],
- # Integer-based embeddings are then re-ranked using the
- # float-based embeddings. Here we just want to retrieve
- # 25 documents.
- query=[-1.234, 0.762, ..., 1.532],
+ # The same query, embedded for the dense retriever.
+ query=dense_embed(query_text),
using="dense",
- limit=25,
+ limit=100,
),
- # Here we just add another 25 documents using the sparse
- # vectors only.
models.Prefetch(
- query=models.SparseVector(
- indices=[125, 9325, 58214],
- values=[-0.164, 0.229, 0.731],
- ),
- using="sparse",
- limit=25,
+ # The same query, embedded for the sparse retriever.
+ query=sparse_embed(query_text),
+ using="bm25",
+ limit=100,
),
],
- # RRF is activated below, so there is no need to specify the
- # query vector here, as fusion is done on the scores of the
- # retrieved documents.
- query=models.FusionQuery(
- fusion=models.Fusion.RRF,
- ),
-)
-```
-
-The second branch could have already been called hybrid, as it combines the results from the dense and sparse vectors
-with fusion. However, nothing stops us from building even more complex search pipelines.
-
-Here is how the target call to the Query API would look like in Python:
-
-
-```python
-client.query_points(
- "my-collection",
- prefetch=[
- matryoshka_prefetch,
- sparse_dense_rrf_prefetch,
- ],
- # Finally rerank the results with the late interaction model. It only
- # considers the documents retrieved by all the prefetch operations above.
- # Return 10 final results.
- query=[
- [1.928, -0.654, ..., 0.213],
- [-1.197, 0.583, ..., 1.901],
- ...,
- [0.112, -1.473, ..., 1.786],
- ],
- using="late-interaction",
- with_payload=False,
+ query=models.FusionQuery(fusion=models.Fusion.RRF),
+ # Results returned to the caller.
limit=10,
)
```
-The options are endless, the new Query API gives you the flexibility to experiment with different setups. **You
-rarely need to build such a complex search pipeline**, but it's good to know that you can do that if needed.
+`using` selects the named vector for each prefetch, and `FusionQuery` merges the two lists. Both prefetch limits start at 100.
-
+## Measure Whether It Helps
-## Lessons learned: multi-vector representations
+Across five public datasets, default RRF beat the stronger individual retriever on four.
-Many of you have already started building hybrid search systems and reached out to us with questions and feedback.
-We've seen many different approaches, however one recurring idea was to utilize **multi-vector representations with
-ColBERT-style models as a reranking step**, after retrieving candidates with single-vector dense and/or sparse methods.
-This reflects the latest trends in the field, as single-vector methods are still the most efficient, but multivectors
-capture the nuances of the text better.
+
-
+_DBPedia-entity is the exception: fusion scores lower than dense retrieval._
-Assuming you never use late interaction models for retrieval alone, but only for reranking, this setup comes with a
-hidden cost. By default, each configured dense vector of the collection will have a corresponding HNSW graph created.
-Even, if it is a multi-vector.
+Run the same labeled queries with dense retrieval, sparse retrieval, and fusion. Keep the models and candidate limits unchanged. Score each run with `nDCG@10`, which gives more credit to relevant documents near the top.
-```python
-from qdrant_client import QdrantClient, models
+First, check whether fusion beats both retrievers. Review the queries where their rankings differ, then see whether the wins and losses cluster around important query types in your workload.
-client = QdrantClient(...)
-client.create_collection(
- collection_name="my-collection",
- vectors_config={
- "dense": models.VectorParams(...),
- "late-interaction": models.VectorParams(
- size=128,
- distance=models.Distance.COSINE,
- multivector_config=models.MultiVectorConfig(
- comparator=models.MultiVectorComparator.MAX_SIM
- ),
- )
- },
- sparse_vectors_config={
- "sparse": models.SparseVectorParams(...)
- },
-)
-```
+If one retriever finds relevant results that fusion ranks too low, tune the fusion method or weights. [How to Tune Hybrid Search](/articles/how-to-tune-hybrid-search/) covers those settings. If both retrievers miss a result, fusion has no candidate to promote.
-Reranking will never use the created graph, as all the candidates are already retrieved. Multi-vector ranking will only
-be applied to the candidates retrieved by the previous steps, so no search operation is needed. HNSW becomes redundant
-while still the indexing process has to be performed, and in that case, it will be quite heavy. ColBERT-like models
-create hundreds of embeddings for each document, so the overhead is significant. **To avoid it, you can disable the HNSW
-graph creation for this kind of model**:
+Recheck the winning setup on held-out queries. [Building a labeled set](/articles/before-tuning-a-qdrant-collection/) covers query selection and held-out evaluation.
-```python
-client.create_collection(
- collection_name="my-collection",
- vectors_config={
- "dense": models.VectorParams(...),
- "late-interaction": models.VectorParams(
- size=128,
- distance=models.Distance.COSINE,
- multivector_config=models.MultiVectorConfig(
- comparator=models.MultiVectorComparator.MAX_SIM
- ),
- hnsw_config=models.HnswConfigDiff(
- m=0, # Disable HNSW graph creation
- ),
- )
- },
- sparse_vectors_config={
- "sparse": models.SparseVectorParams(...)
- },
-)
-```
+Dense retrieval may already cover much of a natural-language-only workload, but query shape alone cannot tell you whether hybrid search will help.
-You won't notice any difference in the search performance, but the use of resources will be significantly lower when you
-upload the embeddings to the collection.
+Keep the sparse retriever when the relevance gain justifies its measured indexing and latency cost.
-## Some anecdotal observations
+## What to Test Next
-Neither of the algorithms performs best in all cases. In some cases, keyword-based search
-will be the winner and vice-versa. The following table shows some interesting examples we could find in the
-[WANDS](https://github.com/wayfair/WANDS) dataset during experimentation:
-
-
-
-
Query
-
BM25 Search
-
Vector Search
-
-
-
-
cybersport desk
-
desk ❌
-
gaming desk ✅
-
-
-
plates for icecream
-
"eat" plates on wood wall décor ❌
-
alicyn 8.5 '' melamine dessert plate ✅
-
-
-
kitchen table with a thick board
-
craft kitchen acacia wood cutting board ❌
-
industrial solid wood dining table ✅
-
-
-
wooden bedside table
-
30 '' bedside table lamp ❌
-
portable bedside end table ✅
-
-
-
-
-
-Also examples where keyword-based search did better:
-
-
-
-
Query
-
BM25 Search
-
Vector Search
-
-
-
-
computer chair
-
vibrant computer task chair ✅
-
office chair ❌
-
-
-
64.2 inch console table
-
cervantez 64.2 '' console table ✅
-
69.5 '' console table ❌
-
-
-
-
-## Try the New Query API in Qdrant 1.10
-
-The new Query API introduced in Qdrant 1.10 is a game-changer for building hybrid search systems. You don't need any
-additional services to combine the results from different search methods, and you can even create more complex pipelines
-and serve them directly from Qdrant.
-
-Our webinar on *Building the Ultimate Hybrid Search* takes you through the process of building a hybrid search system
-with Qdrant Query API. If you missed it, you can [watch the recording](https://www.youtube.com/watch?v=LAZOxqzceEU), or
-[check the notebooks](https://github.com/qdrant/workshop-ultimate-hybrid-search).
-
-
-
-If you have any questions or need help with building your hybrid search system, don't hesitate to reach out to us on
-[Discord](https://qdrant.to/discord).
+- **Add more stages.** [Multi-stage queries](/documentation/search/hybrid-queries/#multi-stage-queries) retrieve with a cheap representation and rescore with an expensive one. Cross-encoder reranking puts the query and chunk into a model together. [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) compares that approach with a tuned first stage.
+- **Tune what you have.** The fusion method, the RRF constant, and the per-retriever weights can all move relevance without adding a stage. [How to Tune Hybrid Search](/articles/how-to-tune-hybrid-search/) measures each one across the same five datasets.
diff --git a/qdrant-landing/content/articles/indexing-optimization.md b/qdrant-landing/content/articles/indexing-optimization.md
index 9349ca734..921a34892 100644
--- a/qdrant-landing/content/articles/indexing-optimization.md
+++ b/qdrant-landing/content/articles/indexing-optimization.md
@@ -8,6 +8,8 @@ weight: 30
author: Sabrina Aquino
date: 2025-02-13T00:00:00.000Z
category: production-ops
+hideFromList: true
+draft: true
---
# Optimizing Memory Consumption During Bulk Uploads
diff --git a/qdrant-landing/content/articles/langchain-integration.md b/qdrant-landing/content/articles/langchain-integration.md
index d0187c03e..86bebc109 100644
--- a/qdrant-landing/content/articles/langchain-integration.md
+++ b/qdrant-landing/content/articles/langchain-integration.md
@@ -1,14 +1,13 @@
---
-title: "Using LangChain for Question Answering with Qdrant"
-short_description: "Large Language Models might be developed fast with modern tool. Here is how!"
-description: "We combined LangChain, a pre-trained LLM from OpenAI, SentenceTransformers & Qdrant to create a question answering system with just a few lines of code. Learn more!"
+title: "Question Answering with LangChain and Qdrant"
+short_description: "Build a retrieval-augmented question answering pipeline with just a few lines of code."
+description: "We combined LangChain, a modern chat model like Claude or GPT, FastEmbed & Qdrant to create a question answering system with just a few lines of code. Learn more!"
social_preview_image: /articles_data/langchain-integration/preview/social_preview.jpg
-small_preview_image: /articles_data/langchain-integration/chain.svg
preview_dir: /articles_data/langchain-integration/preview
weight: 40
-author: Kacper Łukawski
+author: Kacper Łukawski and Manas Chopra
author_link: https://medium.com/@lukawskikacper
-date: 2023-01-31T10:53:20+01:00
+date: 2026-07-30T10:00:00+03:00
draft: false
keywords:
- vector search
@@ -17,35 +16,44 @@ keywords:
- large language models
- question answering
- openai
+ - anthropic
+ - claude
+ - fastembed
- embeddings
category: demos-and-tutorials
---
-# Streamlining Question Answering: Simplifying Integration with LangChain and Qdrant
+
+ Follow along in Colab:
+
+
-Building applications with Large Language Models doesn't have to be complicated. A lot has been going on recently to simplify the development,
-so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://langchain.readthedocs.io)
+Building applications with Large Language Models doesn't have to be complicated. A lot has been going on recently to simplify the development,
+so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://docs.langchain.com/oss/python/langchain/overview)
provides unified interfaces to different libraries, so you can avoid writing boilerplate code and focus on the value you want to bring.
## Why Use Qdrant for Question Answering with LangChain?
-It has been reported millions of times recently, but let's say that again. ChatGPT-like models struggle with generating factual statements if no context
-is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus, it is better to provide some facts we
-know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give a comprehensive answer. [Vector database,
-such as Qdrant](https://qdrant.tech/), is of great help here, as their ability to perform a [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid
-documents, so they can be provided into the LLM. That's also one of the **chains** implemented in [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which is called `VectorDBQA`. And Qdrant got
-integrated with the library, so it might be used to build it effortlessly.
+It has been reported millions of times, but let's say it again. Modern LLMs, whether that's Claude, GPT, or any other chat model, still struggle to
+generate factual statements if no context is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus,
+it is better to provide some facts we know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give
+a comprehensive answer. A [vector search engine, such as Qdrant](https://qdrant.tech/), is of great help here, as its ability to perform a
+[semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid
+documents, so they can be provided into the LLM. This pattern is commonly known as retrieval-augmented generation, and it is one of the core building
+blocks of [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which got Qdrant integrated as a first-class vector store, so it might be
+used to build such pipelines effortlessly.
### The Two-Model Approach
Surprisingly enough, there will be two models required to set things up. First of all, we need an embedding model that will convert the set of facts into
-vectors, and store those into Qdrant. That's an identical process to any other semantic search application. We're going to use one of the
-`SentenceTransformers` models, so it can be hosted locally. The embeddings created by that model will be put into Qdrant and used to retrieve the most
-similar documents, given the query.
+vectors, and store those into Qdrant. That's an identical process to any other semantic search application. We're going to use
+[FastEmbed](https://qdrant.tech/articles/fastembed/), Qdrant's own lightweight embedding library, so it can be hosted locally without pulling in a full
+PyTorch or TensorFlow stack. The embeddings created by that model will be put into Qdrant and used to retrieve the most similar documents, given the query.
-However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all
-of them into a single text. Then, we build a prompt to the LLM (in our case [OpenAI](https://openai.com/)), including those documents as a context, of course together with the
-question asked. So the input to the LLM looks like the following:
+However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all
+of them into a single text. Then, we build a prompt to the chat model (in our examples below, either [Anthropic's Claude](https://www.anthropic.com/claude)
+or [OpenAI's GPT](https://openai.com/)), including those documents as a context, of course together with the question asked. So the input to the LLM
+looks like the following:
```text
Use the following pieces of context to answer the question at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer.
@@ -56,54 +64,158 @@ Question: How much is 2 + 2?
Helpful Answer:
```
-There might be several context documents combined, and it is solely up to LLM to choose the right piece of content. But our expectation is, the model should
-respond with just `4`.
+There might be several context documents combined, and it is solely up to the LLM to choose the right piece of content. But our expectation is, the model
+should respond with just `4`.
-## Why do we need two different models?
+## Why do we need two different models?
Both solve some different tasks. The first model performs feature extraction, by converting the text into vectors, while
-the second one helps in text generation or summarization. Disclaimer: This is not the only way to solve that task with LangChain. Such a chain is called `stuff`
-in the library nomenclature.
+the second one helps in text generation or summarization. Disclaimer: this is not the only way to solve that task with LangChain. Since we simply stuff
+all the retrieved documents into a single prompt, this pattern is often called a **stuff** chain.

-Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few lines
-of code, thanks to the recent integration with [Qdrant](https://qdrant.tech/). We're not even going to work directly with `QdrantClient`, as everything is already done in the background
-by LangChain. If you want to get into the source code right away, all the processing is available as a
-[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing).
+Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few
+lines of code, thanks to the integration with [Qdrant](https://qdrant.tech/). We're not even going to work directly with `QdrantClient`, as everything is
+already done in the background by LangChain.
## How to Implement Question Answering with LangChain and Qdrant
### Step 1: Configuration
-A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant Cloud](https://cloud.qdrant.io),
-so we need an API key. The same is for OpenAI - the API key has to be obtained from their website.
+Before anything else, install the packages this pipeline touches - LangChain's Qdrant integration, FastEmbed, the `datasets` library for loading Natural
+Questions, and whichever chat model provider you'd like to call:
-
+```shell
+pip install langchain langchain-qdrant fastembed datasets langchain-anthropic langchain-openai
+```
+
+A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant
+Cloud](https://cloud.qdrant.io), so we need a URL and an API key. On the generation side, you can plug in whichever chat model you prefer - an API key
+from [Anthropic](https://console.anthropic.com/) or [OpenAI](https://platform.openai.com/) is all you need, since LangChain exposes the same interface for
+both.
+
+```python
+import os
+
+os.environ["QDRANT_URL"] = "https://xxxxxx-xxxxxx.xxx.aws.cloud.qdrant.io"
+os.environ["QDRANT_API_KEY"] = ""
+
+# Pick whichever provider you'd like to use for generating the answers
+os.environ["ANTHROPIC_API_KEY"] = "" # for Claude
+os.environ["OPENAI_API_KEY"] = "" # for GPT
+```
### Step 2: Building the knowledge base
-We also need some facts from which the answers will be generated. There is plenty of public datasets available, and
-[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them. It consists of the whole HTML content of the websites they were
-scraped from. That means we need some preprocessing to extract plain text content. As a result, we’re going to have two lists of strings - one for questions and
-the other one for the answers.
+We also need some facts from which the answers will be generated. There is plenty of public datasets available, and
+[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them - a collection of real Google search queries paired
+with the relevant passage from Wikipedia that answers them. Rather than parsing the raw, HTML-heavy release ourselves, we can pull the
+already-cleaned `query`/`answer` pairs published on the Hugging Face Hub, which gets us two lists of strings - one for questions and the other one for
+the answers - in a single call.
-The answers have to be vectorized with the first of our models. The `sentence-transformers/all-mpnet-base-v2` is one of the possibilities, but there are some
-other options available. LangChain will handle that part of the process in a single function call.
+```python
+from datasets import load_dataset
-
+dataset = load_dataset("sentence-transformers/natural-questions", split="train")
+# 100 pairs is enough to experiment with; drop the .select() call entirely to index all 100k+ rows
+dataset = dataset.select(range(100))
-### Step 3: Setting up QA with Qdrant in a loop
+questions = dataset["query"]
+answers = dataset["answer"]
+```
-`VectorDBQA` is a chain that performs the process described above. So it, first of all, loads some facts from Qdrant and then feeds them into OpenAI LLM which
-should analyze them to find the answer to a given question. The only last thing to do before using it is to put things together, also with a single function call.
+The answers have to be vectorized with our embedding model. FastEmbed defaults to
+[`BAAI/bge-small-en-v1.5`](https://huggingface.co/BAAI/bge-small-en-v1.5), a small, quantized model that runs comfortably on CPU, but there are
+[several other models](https://qdrant.github.io/fastembed/examples/Supported_Models/) to pick from. `langchain-community`, which used to ship a
+`FastEmbedEmbeddings` wrapper, is [being sunset](https://github.com/langchain-ai/langchain-community/issues/674), so instead we wrap FastEmbed's
+`TextEmbedding` directly with LangChain's `Embeddings` interface - it's a handful of lines, and it keeps the pipeline free of a deprecated dependency.
+LangChain will still handle vectorizing the documents and creating the Qdrant collection in a single function call.
-
+```python
+from typing import List
+
+from fastembed import TextEmbedding
+from langchain_core.embeddings import Embeddings
+from langchain_qdrant import QdrantVectorStore
+
+
+class FastEmbedEmbeddings(Embeddings):
+ def __init__(self, model_name: str = "BAAI/bge-small-en-v1.5"):
+ self._model = TextEmbedding(model_name=model_name)
+
+ def embed_documents(self, texts: List[str]) -> List[List[float]]:
+ return [vector.tolist() for vector in self._model.embed(texts)]
+
+ def embed_query(self, text: str) -> List[float]:
+ return self.embed_documents([text])[0]
+
+
+embeddings = FastEmbedEmbeddings()
+
+doc_store = QdrantVectorStore.from_texts(
+ answers,
+ embeddings,
+ url=os.environ["QDRANT_URL"],
+ api_key=os.environ["QDRANT_API_KEY"],
+ collection_name="natural-questions",
+)
+```
+
+### Step 3: Setting up the retrieval chain
+
+With the knowledge base in place, the only thing left is to combine the retriever with a chat model. [`init_chat_model`](https://docs.langchain.com/oss/python/langchain/models)
+lets us pass in the name of any supported model - Claude, GPT, or otherwise - without changing the rest of the pipeline. We then compose the retriever, the
+prompt, and the model using LangChain's expression language (LCEL), so the whole chain is defined with a single `|`-piped expression.
+
+```python
+from langchain.chat_models import init_chat_model
+from langchain_core.output_parsers import StrOutputParser
+from langchain_core.prompts import ChatPromptTemplate
+from langchain_core.runnables import RunnablePassthrough
+
+retriever = doc_store.as_retriever()
+
+prompt = ChatPromptTemplate.from_template(
+ """Use the following pieces of context to answer the question at the end. If you don't know the answer, just
+say that you don't know, don't try to make up an answer.
+
+{context}
+
+Question: {question}
+Helpful Answer:"""
+)
+
+def format_docs(docs):
+ return "\n\n".join(doc.page_content for doc in docs)
+
+# Swap the model name for any other provider LangChain supports, e.g. "gpt-5.1"
+llm = init_chat_model("claude-sonnet-4-5", model_provider="anthropic")
+
+chain = (
+ {"context": retriever | format_docs, "question": RunnablePassthrough()}
+ | prompt
+ | llm
+ | StrOutputParser()
+)
+```
## Step 4: Testing out the chain
-And that's it! We can put some queries, and LangChain will perform all the required processing to find the answer in the provided context.
+And that's it! We can put in some queries, and LangChain will perform all the required processing to find the answer in the provided context. Since we
+already have a `questions` list from the dataset, let's just sample a handful of them and see how the chain responds:
-
+```python
+import random
+
+random.seed(76)
+selected_questions = random.choices(questions, k=5)
+for question in selected_questions:
+ print(">", question)
+ print(chain.invoke(question), end="\n\n")
+```
+
+The exact wording will vary depending on which chat model you plug in, but running the chain against the same knowledge base as the original experiment
+produces answers along these lines:
```text
> what kind of music is scott joplin most famous for
@@ -123,7 +235,4 @@ And that's it! We can put some queries, and LangChain will perform all the requi
```
The great thing about such a setup is that the knowledge base might be easily extended with some new facts and those will be included in the prompts
-sent to LLM later on. Of course, assuming their similarity to the given question will be in the top results returned by Qdrant.
-
-If you want to run the chain on your own, the simplest way to reproduce it is to open the
-[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing).
+sent to the LLM later on. Of course, assuming their similarity to the given question will be in the top results returned by Qdrant.
diff --git a/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md b/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md
new file mode 100644
index 000000000..985612589
--- /dev/null
+++ b/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md
@@ -0,0 +1,127 @@
+---
+title: "Memory Tiers in Qdrant: What to Use and When"
+short_description: "A guide to choosing a Qdrant memory tier layout as your collection grows."
+description: "Which Qdrant memory tier layout to use and when, and why, backed by benchmarks."
+social_preview_image: /articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.jpg
+preview_dir: /articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview
+author: Clelia Bertelli
+author_link: https://qdrant.tech
+date: 2026-08-28T10:00:00+02:00
+draft: false
+keywords:
+ - memory tiers
+ - caching
+ - disk
+ - scaling
+ - benchmark
+category: production-ops
+weight: 8
+---
+
+A growing vector collection eventually outgrows the RAM it started with: Qdrant handles that by letting you assign dense vectors, the HNSW graph, quantized vectors, payloads, and payload indexes each to whichever memory tier that structure supports, instead of forcing one RAM-versus-disk trade-off onto the whole collection.
+
+This article will give you practical guidance over which combination of tiers and quantization to reach for at each stage of a collection's growth, and the reasons behind the choice.
+
+## How Qdrant's Memory Tiers Work
+
+RAM is fast and expensive, disk is slow and cheap, and a growing collection outgrows its RAM budget long before it outgrows its disk budget. Qdrant gives you three tiers to manage that trade-off:
+
+- **Pinned.** Lives on the heap and never gets evicted, so access is always fast.
+- **Cached.** Memory-mapped and pre-warmed into the OS page cache at startup, so it starts fast too, but the OS can push it out under memory pressure.
+- **Cold.** Also memory-mapped, without the pre-warming, so the first read of any page comes from disk.
+
+Not every structure supports every tier: dense vectors and payloads can't be pinned, and sparse vector structures follow their own rules. For the full breakdown, see the [memory tiers documentation](/documentation/ops-configuration/memory-tiers/).
+
+
+
+Quantization is one of the main tools that makes the cold and pinned tiers practical at real scale. Compressing vectors to int8 or lower shrinks them enough to fit a small, fixed amount of RAM. A search can then score most candidates against that compressed copy instead of paging in the full-precision ones. Rescoring is on by default only for binary quantization and low-precision TurboQuant.
+
+## Combining Tiers and Quantization
+
+Tiers and quantization combine into a handful of configurations that cover most collections, from a small one that fits entirely in RAM to one too large for RAM at any budget:
+
+- **Full Cached, No Quantization.** Dense vectors and the HNSW graph both cached. The simplest option, and a good default as long as the whole working set fits comfortably in RAM.
+- **Full Cached, With Quantization.** The same, plus a compressed copy of the vectors also cached, overriding Qdrant's default of pinning that copy instead. This keeps two copies of the vector data warm at once, so it only makes sense with RAM to spare for both.
+- **Full Cold, No Quantization.** Dense vectors and the graph both memory-mapped without pre-warming. Minimizes RAM use, but every first touch on a page comes from disk.
+- **Full Cold, With Quantization.** The cold tier, plus a compressed copy that's also left cold, matching Qdrant's own default for that combination. Lower disk-read cost than the unquantized cold tier, since most candidates score against the small compressed copy instead.
+- **Pinned Quantized Vectors.** Dense vectors kept cold, but the compressed copy explicitly pinned in RAM. This is the configuration Qdrant's optimization docs recommend for high-speed search with a low memory footprint, and the one to reach for once RAM headroom becomes the binding constraint.
+
+
+
+Four of these five configurations leave their memory cost up to the OS: how much of that expected footprint actually stays resident depends on what else is competing for RAM. Only pinning turns that footprint into a fixed reservation instead.
+
+## Pin Quantized Vectors for Predictability
+
+Pin the compressed copy once a collection is large enough that RAM becomes the binding constraint, not just raw speed. That fixed reservation is what keeps pinning safe as a collection keeps growing, well past the point where the other configurations start running out of memory headroom.
+
+
+
+Pinning tends to land in that efficient corner: it matches the fastest configuration's search speed while using a fraction of its memory footprint. The configurations that leave their footprint up to the OS spike toward the cluster's memory ceiling as a collection grows, while the pinned one's footprint barely moves.
+
+
+
+On a collection small enough to fit entirely in the cached tier with room to spare, skip pinning: it adds configuration for no speed advantage, since caching the full-precision vectors matches pinning on latency as long as everything actually fits.
+
+## Use The Cold Tier With Quantization
+
+If disk footpring matters more than just raw speed, consider adding quantization before fully committing to the cold tier. Every **first touch on a cold, unquantized structure comes from disk**, and a single HNSW traversal touches enough pages that a rare query can stretch far past its typical latency.
+
+Giving the search a small compressed copy to score against, instead of paging full-precision vectors in from a cold file, removes most of the tail-latency risk even before anything gets pinned.
+
+
+
+A disk page holds only a handful of full-precision vectors, so most candidates a traversal touches need their own read. A compressed copy packs many more vectors onto the same page, so one read serves far more of the candidates a query needs.
+
+Most queries against an unquantized cold tier still come back fine: the risk is in the tail, where a rare hop lands on a page that isn't already cached and pays for that read in full, a condition the quantized configurations don't reproduce.
+
+
+
+## When To Cache Everything
+
+Don't cache a compressed copy alongside full-precision vectors unless RAM has room for both. It doubles the resident working set instead of shrinking it, since Qdrant's default is to pin that compressed copy rather than cache it.
+
+Caching the full-precision vectors alone is fast and simple as long as RAM has room for it. The risk shows up once the working set stops fitting: a configuration that performed close to the fastest option at a smaller scale can fall to nearly the worst as its resident data approaches the cluster's memory budget.
+
+
+
+The failure isn't the tier logic breaking, but the cluster running out of memory to keep that much data warm at once: a sizing problem rather than a caching one, but the cached tier is uniquely exposed to it.
+
+In these cases, the latency can increase by multiple times, and its worst-case queries stretch far past normal, as its working set is very close to the cluster's memory ceiling.
+
+
+
+## HNSW Inline Storage and Disk Space
+
+Test [HNSW inline storage](/documentation/ops-optimization/optimize/#inline-storage-in-hnsw-index) with quantization at your real vector dimensionality and graph connectivity before relying on it in production. Qdrant's inline HNSW storage removes a random-access read by copying vector data directly into the graph file: a full-precision copy per point, plus a compressed copy for every edge into it.
+
+That second cost scales with edge count and vector size, not point count, so it can grow far faster than the collection does.
+
+
+
+Combining inline storage with quantization can push on-disk graph size to many times larger than the same graph without inline storage. Dense, high-dimensional graphs push the per-edge cost further than sparse, low-dimensional ones, so treat any single multiplier you measure as workload-specific rather than universal. A smaller compressed vector size shrinks it: fewer bytes per edge means less extra disk, though it won't remove the problem entirely.
+
+
+
+## Treat RAM as a Limiting Factor
+
+Size each configuration against how much of the cluster's RAM its working set occupies right now, not how many points the collection holds or the headroom you had at initial setup. A single configuration can look fine, fail, and recover again, purely because the ratio between dataset size and available RAM shifts underneath it.
+
+
+
+A story built only on point count, where more data always means a worse tail, doesn't hold up: a caching configuration that looks competitive at a smaller scale can fall apart at a bigger one, once its working set comes near the cluster's entire memory budget, then look fine again as soon as the cluster's own RAM budget grows to match.
+
+
+
+## Takeaways
+
+- Pin the compressed copy once RAM headroom, not raw speed, becomes the binding constraint. Skip it on collections small enough to fit entirely in the cached tier.
+- Add quantization before going cold if disk footprint matters more than raw speed; never leave dense vectors cold and unquantized in a latency-sensitive workload.
+- Don't cache a compressed copy alongside full-precision vectors unless the cluster has RAM to spare for both.
+- Test HNSW inline storage with quantization at your real vector dimensionality and graph connectivity before trusting it in production.
+- Re-check RAM headroom against the working set at every scaling step, not just at initial setup.
+
+## Adjacent Work
+
+- [Memory tiers documentation](/documentation/ops-configuration/memory-tiers/): the full set of tier and quantization options per structure.
+- [Storage documentation](/documentation/manage-data/storage/): how collections, segments, and storage structures fit together on disk.
+- [qdrant-labs/memory-tiers-explained](https://github.com/qdrant-labs/memory-tiers-explained): the benchmark code and raw results behind the guidance in this piece.
diff --git a/qdrant-landing/content/articles/multitenancy.md b/qdrant-landing/content/articles/multitenancy.md
index d4e0bed90..28b021eea 100644
--- a/qdrant-landing/content/articles/multitenancy.md
+++ b/qdrant-landing/content/articles/multitenancy.md
@@ -19,14 +19,14 @@ category: production-ops
# Scaling Your Machine Learning Setup: The Power of Multitenancy and Custom Sharding in Qdrant
-We are seeing the topics of [multitenancy](/documentation/manage-data/multitenancy/) and [distributed deployment](/documentation/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup.
+We are seeing the topics of [multitenancy](/documentation/manage-data/multitenancy/) and [distributed deployment](/documentation/scaling/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup.
Whether you are building a bank fraud-detection system, [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) for e-commerce, or services for the federal government - you will need to leverage a multitenant architecture to scale your product.
In the world of SaaS and enterprise apps, this setup is the norm. It will considerably increase your application's performance and lower your hosting costs.
## Multitenancy & custom sharding with Qdrant
-We have developed two major features just for this. __You can now scale a single Qdrant cluster and support all of your customers worldwide.__ Under [multitenancy](/documentation/manage-data/multitenancy/), each customer's data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer's access. This is called [custom sharding](/documentation/distributed_deployment/#user-defined-sharding).
+We have developed two major features just for this. __You can now scale a single Qdrant cluster and support all of your customers worldwide.__ Under [multitenancy](/documentation/manage-data/multitenancy/), each customer's data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer's access. This is called [custom sharding](/documentation/scaling/distributed_deployment/#user-defined-sharding).
Combining these two will result in an efficiently-partitioned architecture that further leverages the convenience of a single Qdrant cluster. This article will briefly explain the benefits and show how you can get started using both features.
@@ -41,7 +41,7 @@ Qdrant is built to excel in a single collection with a vast number of tenants. Y
## Sharding your database
-With Qdrant, you can also specify a shard for each vector individually. This feature is useful if you want to [control where your data is kept in the cluster](/documentation/distributed_deployment/#sharding). For example, one set of vectors can be assigned to one shard on its own node, while another set can be on a completely different node.
+With Qdrant, you can also specify a shard for each vector individually. This feature is useful if you want to [control where your data is kept in the cluster](/documentation/scaling/distributed_deployment/#sharding). For example, one set of vectors can be assigned to one shard on its own node, while another set can be on a completely different node.
During vector search, your operations will be able to hit only the subset of shards they actually need. In massive-scale deployments, __this can significantly improve the performance of operations that do not require the whole collection to be scanned__.
@@ -49,7 +49,7 @@ This works in the other direction as well. Whenever you search for something, yo
### Common use cases
-A clear use-case for this feature is managing a multitenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. Sharding solves the problem of region-based data placement, whereby certain data needs to be kept within specific locations. To do this, however, you will need to [move your shards between nodes](/documentation/distributed_deployment/#moving-shards).
+A clear use-case for this feature is managing a multitenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. Sharding solves the problem of region-based data placement, whereby certain data needs to be kept within specific locations. To do this, however, you will need to [move your shards between nodes](/documentation/scaling/distributed_deployment/#moving-shards).
**Figure 2:** Users can both upsert and query shards that are relevant to them, all within the same collection. Regional sharding can help avoid cross-continental traffic.

@@ -70,7 +70,8 @@ When creating a collection, you will need to configure user-defined sharding. Th
```python
client.create_collection(
collection_name="{tenant_data}",
- shard_number=2,
+ # number of physical shards per shard key, not the number of shard keys
+ shard_number=1,
sharding_method=models.ShardingMethod.CUSTOM,
# ... other collection parameters
)
@@ -79,7 +80,7 @@ client.create_shard_key("{tenant_data}", "germany")
```
In this example, your cluster is divided between Germany and Canada. Canadian and German law differ when it comes to international data transfer. Let's say you are creating a RAG application that supports the healthcare industry. Your Canadian customer data will have to be clearly separated for compliance purposes from your German customer.
-Even though it is part of the same collection, data from each shard is isolated from other shards and can be retrieved as such. For additional examples on shards and retrieval, consult [Distributed Deployments](/documentation/distributed_deployment/) documentation and [Qdrant Client specification](https://python-client.qdrant.tech).
+Even though it is part of the same collection, data from each shard is isolated from other shards and can be retrieved as such. For additional examples on shards and retrieval, consult [Distributed Deployments](/documentation/scaling/distributed_deployment/) documentation and [Qdrant Client specification](https://python-client.qdrant.tech).
## Configure a multitenant setup for users
@@ -126,7 +127,7 @@ client.upsert(
The access control setup is completed as you specify the criteria for data retrieval. When searching for vectors, you need to use a `query_filter` along with `group_id` to filter vectors for each user.
```python
-client.search(
+client.query_points(
collection_name="{tenant_data}",
query_filter=models.Filter(
must=[
@@ -138,7 +139,7 @@ client.search(
),
]
),
- query_vector=[0.1, 0.1, 0.9],
+ query=[0.1, 0.1, 0.9],
limit=10,
)
```
@@ -194,4 +195,3 @@ Get support or share ideas in our [Discord](https://qdrant.to/discord) community
-
diff --git a/qdrant-landing/content/articles/neural-search-tutorial.md b/qdrant-landing/content/articles/neural-search-tutorial.md
index 2ce9d8e93..8b57575b0 100644
--- a/qdrant-landing/content/articles/neural-search-tutorial.md
+++ b/qdrant-landing/content/articles/neural-search-tutorial.md
@@ -259,15 +259,15 @@ The search function looks as simple as possible:
vector = self.model.encode(text).tolist()
# Use `vector` for search for closest vectors in the collection
- search_result = self.qdrant_client.search(
+ search_result = self.qdrant_client.query_points(
collection_name=self.collection_name,
- query_vector=vector,
+ query=vector,
query_filter=None, # We don't want any filters for now
- top=5 # 5 the most closest results is enough
+ limit=5 # 5 the most closest results is enough
)
# `search_result` contains found vector ids with similarity scores along with the stored payload
# In this function we are interested in payload only
- payloads = [hit.payload for hit in search_result]
+ payloads = [hit.payload for hit in search_result.points]
return payloads
```
@@ -291,11 +291,11 @@ from qdrant_client.models import Filter
}]
})
- search_result = self.qdrant_client.search(
+ search_result = self.qdrant_client.query_points(
collection_name=self.collection_name,
- query_vector=vector,
+ query=vector,
query_filter=city_filter,
- top=5
+ limit=5
)
...
diff --git a/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md b/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md
index 5e647c5f9..25f970784 100644
--- a/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md
+++ b/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md
@@ -190,13 +190,13 @@ was present in the first *k* results.
k_max = 10
answer_positions = []
for embedding, pubid in tqdm(zip(question_response.embeddings, ids)):
- response = qdrant_client.search(
+ response = qdrant_client.query_points(
collection_name="pubmed_qa",
- query_vector=embedding,
+ query=embedding,
limit=k_max,
)
- answer_ids = [record.id for record in response]
+ answer_ids = [record.id for record in response.points]
if pubid in answer_ids:
answer_positions.append(answer_ids.index(pubid))
else:
diff --git a/qdrant-landing/content/articles/qdrant-1.7.x.md b/qdrant-landing/content/articles/qdrant-1.7.x.md
index da20cc9ad..a237dbe98 100644
--- a/qdrant-landing/content/articles/qdrant-1.7.x.md
+++ b/qdrant-landing/content/articles/qdrant-1.7.x.md
@@ -92,7 +92,7 @@ POST /collections/my_collection/points/search
}
```
-If you want to know more about the user-defined sharding, please refer to the [sharding documentation](/documentation/distributed_deployment/#sharding).
+If you want to know more about the user-defined sharding, please refer to the [sharding documentation](/documentation/scaling/distributed_deployment/#sharding).
### Snapshot-based shard transfer
@@ -101,7 +101,7 @@ That's a really more in depth technical improvement for the distributed mode use
Moving shards is required for dynamical scaling of the cluster. Your data can migrate between nodes, and the way you move it is crucial for the performance of the whole system. The good old `stream_records` method (still the default one) transmits all the records between the machines and indexes them on the target node.
In the case of moving the shard, it's necessary to recreate the HNSW index each time. However, with the introduction of the new `snapshot` approach, the snapshot itself, inclusive of all data and potentially quantized content, is transferred to the target node. This comprehensive snapshot includes the entire index, enabling the target node to seamlessly load it and promptly begin handling requests without the need for index recreation.
-There are multiple scenarios in which you may prefer one over the other. Please check out the docs of the [shard transfer method](/documentation/distributed_deployment/#shard-transfer-method) for more details and head-to-head comparison. As for now, the old `stream_records` method is still the default one, but we may decide to change it in the future.
+There are multiple scenarios in which you may prefer one over the other. Please check out the docs of the [shard transfer method](/documentation/scaling/distributed_deployment/#shard-transfer-method) for more details and head-to-head comparison. As for now, the old `stream_records` method is still the default one, but we may decide to change it in the future.
## Minor improvements
diff --git a/qdrant-landing/content/articles/rag-is-dead.md b/qdrant-landing/content/articles/rag-is-dead.md
index 0ea957514..7df6041e2 100644
--- a/qdrant-landing/content/articles/rag-is-dead.md
+++ b/qdrant-landing/content/articles/rag-is-dead.md
@@ -1,14 +1,14 @@
---
title: "Is RAG Dead? Why Long Context Windows Don't Replace RAG"
-short_description: Learn how Qdrant’s vector database enhances enterprise AI with superior accuracy and cost-effectiveness.
-description: Uncover the necessity of vector databases for RAG and learn how Qdrant's vector database empowers enterprise AI with unmatched accuracy and cost-effectiveness.
+short_description: Learn how Qdrant enhances enterprise AI with superior accuracy and cost-effectiveness.
+description: Uncover the necessity of vector search for RAG and learn how Qdrant empowers enterprise AI with unmatched accuracy and cost-effectiveness.
social_preview_image: /articles_data/rag-is-dead/preview/social_preview.jpg
small_preview_image: /articles_data/rag-is-dead/icon.svg
preview_dir: /articles_data/rag-is-dead/preview
weight: 60
-author: David Myriel
+author: David Myriel & Chadha Sridi
author_link: https://github.com/davidmyriel
-date: 2024-02-27T00:00:00.000Z
+date: 2026-08-04T00:00:00.000Z
draft: false
keywords:
- vector database
@@ -18,7 +18,7 @@ keywords:
category: core-concepts
---
-# Is RAG Dead? The Role of Vector Databases in AI Efficiency and Vector Search
+# Is RAG Dead? The Role of Vector Search in AI Efficiency
When Anthropic came out with a context window of 100K tokens, they said: “*[Vector search](https://qdrant.tech/solutions/) is dead. LLMs are getting more accurate and won’t need RAG anymore.*”
@@ -36,7 +36,7 @@ The community is already stress testing Gemini 1.5:
This is not surprising. LLMs require massive amounts of compute and memory to run. To cite Grant, running such a model by itself “would deplete a small coal mine to generate each completion”. Also, who is waiting 30 seconds for a response?
-## Context stuffing is not the solution
+## Context Stuffing Is Not the Solution
> Relying on context is expensive, and it doesn’t improve response quality in real-world applications. Retrieval based on [vector search](https://qdrant.tech/solutions/) offers much higher precision.
@@ -44,51 +44,54 @@ If you solely rely on an [LLM](https://qdrant.tech/articles/what-is-rag-in-ai/)
A large context window makes it harder to focus on relevant information. This increases the risk of errors or hallucinations in its responses.
+Large context windows do not guarantee that an LLM will use all available information effectively. The [Lost in the Middle paper](https://arxiv.org/abs/2307.03172) demonstrated that LLMs do not attend uniformly across long contexts: information placed in the middle of the context is often underutilized, causing performance degradation when important facts are placed away from the beginning (primacy bias) or end (recency bias) of the context. Increasing the context window does not solve the problem of finding and using the right information.
+
+
Google found Gemini 1.5 significantly more accurate than GPT-4 at shorter context lengths and “a very small decrease in recall towards 1M tokens”. The recall is still below 0.8.

-We don’t think 60-80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information.
+We don’t think 60 to 80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information.
-> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector database](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all.
+> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector search engine](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all.
Qdrant’s benchmark results are strongly in favor of accuracy and efficiency. We recommend that you consider them before deciding that an LLM is enough. Take a look at our [open-source benchmark reports](/benchmarks/) and [try out the tests](https://github.com/qdrant/vector-db-benchmark) yourself.
-## Vector search in compound systems
+## Vector Search in Compound Systems
The future of AI lies in careful system engineering. As per [Zaharia et al.](https://bair.berkeley.edu/blog/2024/02/18/compound-ai-systems/), results from Databricks find that “60% of LLM applications use some form of RAG, while 30% use multi-step chains.”
Even Gemini 1.5 demonstrates the need for a complex strategy. When looking at [Google’s MMLU Benchmark](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf), the model was called 32 times to reach a score of 90.0% accuracy. This shows us that even a basic compound arrangement is superior to monolithic models.
-As a retrieval system, a [vector database](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run.
+As a retrieval system, a [vector search engine](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run.
> The key advantage of RAG is that it allows an LLM to pull in real-time information from up-to-date internal and external knowledge sources, making it more dynamic and adaptable to new information. - Oliver Molander, CEO of IMAGINAI
>
-## Qdrant scales to enterprise RAG scenarios
+## Qdrant Scales to Enterprise RAG Scenarios
-People still don’t understand the economic benefit of vector databases. Why would a large corporate AI system need a standalone vector database like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether.
+People still don’t understand the economic benefit of vector search. Why would a large corporate AI system need a standalone vector search engine like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether.
**How much would all of this cost?**
-If you are running a RAG solution in an enterprise environment with petabytes of private data, your compute bill will be unimaginable. Let's assume 1 cent per 1K input tokens (which is the current GPT-4 Turbo pricing). Whatever you are doing, every time you go 100 thousand tokens deep, it will cost you $1.
+If you are running a RAG solution in an enterprise environment with petabytes of private data, your compute bill will be unimaginable. Let's assume \\$5 per million input tokens, roughly the pricing of a frontier production model such as GPT-5.6 Sol or Claude Opus. Every time you send 200,000 input tokens, it costs about \\$1 before generating a single output token. At enterprise scale, those costs add up quickly.
That’s a buck a question.
-> According to our estimations, vector search queries are **at least** 100 million times cheaper than queries made by LLMs.
+> Vector search queries are orders of magnitude cheaper than queries made by LLMs.
-Conversely, the only up-front investment with vector databases is the indexing (which requires more compute). After this step, everything else is a breeze. Once setup, Qdrant easily scales via [features like Multitenancy and Sharding](/articles/multitenancy/). This lets you scale up your reliance on the vector retrieval process and minimize your use of the compute-heavy LLMs. As an optimization measure, Qdrant is irreplaceable.
+Conversely, the only up-front investment with vector search engines is the indexing (which requires more compute). After this step, everything else is a breeze. Once setup, Qdrant easily scales via [features like Multitenancy and Sharding](/articles/multitenancy/). This lets you scale up your reliance on the vector retrieval process and minimize your use of the compute-heavy LLMs. As an optimization measure, Qdrant is irreplaceable.
Julien Simon from HuggingFace says it best:
> RAG is not a workaround for limited context size. For mission-critical enterprise use cases, RAG is a way to leverage high-value, proprietary company knowledge that will never be found in public datasets used for LLM training. At the moment, the best place to index and query this knowledge is some sort of vector index. In addition, RAG downgrades the LLM to a writing assistant. Since built-in knowledge becomes much less important, a nice small 7B open-source model usually does the trick at a fraction of the cost of a huge generic model.
-## Get superior accuracy with Qdrant's vector database
+## Get Superior Accuracy with Qdrant
As LLMs continue to require enormous computing power, users will need to leverage vector search and [RAG](https://qdrant.tech/rag/rag-evaluation-guide/).
-Our customers remind us of this fact every day. As a product, [our vector database](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy.
+Our customers remind us of this fact every day. As a product, [our vector search engine](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy.
We want to keep Qdrant compact, efficient and with a focused purpose. This purpose is to empower our customers to use it however they see fit.
diff --git a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md
index 5d1338437..fd538736e 100755
--- a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md
+++ b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md
@@ -9,7 +9,7 @@ weight: 30
author: Atita Arora
author_link: https://github.com/atarora
date: 2024-06-12T00:00:00.000Z
-draft: false
+draft: true
keywords:
- vector database
- vector search
diff --git a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md
index 67ca78f95..dcf810f3c 100644
--- a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md
+++ b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md
@@ -11,7 +11,7 @@ date: 2026-03-09T00:00:00.000Z
category: mastering-search
---
-*This is Part 1 of a 5-part series on fine-tuning sparse embeddings for e-commerce search. We'll go from "why bother?" to a production system that beats BM25 by 29%.*
+*This is Part 1 of a 5-part series on fine-tuning sparse embeddings for e-commerce search. We'll go from "why bother?" to a production system that beats BM25 by 28%.*
**Series:**
- Part 1: Why Sparse Embeddings Beat BM25 (here)
@@ -26,7 +26,7 @@ Search "iPhone 15 Pro Max 256GB" on a dense embedding system and it happily retu

-This is the gap that sparse embeddings fill. And with fine-tuning, they fill it dramatically well - we achieved a **29% improvement over BM25** on Amazon's ESCI dataset, one of the largest public e-commerce search benchmarks.
+This is the gap that sparse embeddings fill. And with fine-tuning, they fill it dramatically well - we achieved a **28% improvement over BM25** on Amazon's ESCI dataset, one of the largest public e-commerce search benchmarks.
In this series, we'll build the entire system: data loading, GPU training on Modal, evaluation with Qdrant, and hard negative mining. The [full code is on GitHub](https://github.com/qdrant-labs/finetune-ecommerce-search) and the [fine-tuned models are on HuggingFace](https://huggingface.co/Qdrant/splade-ecommerce-esci). If you want to skip the walkthrough and fine-tune on your own data, the [`sparse-finetune`](https://github.com/qdrant/sparse-finetune) CLI runs the entire pipeline with one command. But first, let's understand why sparse embeddings are the right tool for e-commerce search.
@@ -169,7 +169,7 @@ Over the next four articles, we'll walk through the full pipeline:
- [**Part 5: From Research to Product**](/articles/sparse-embeddings-ecommerce-part-5/) - An open-source CLI and web dashboard that runs the entire fine-tuning pipeline with a single command.
-The end result: a fine-tuned SPLADE model that achieves **nDCG@10 of 0.388** on Amazon ESCI, compared to **0.301** for BM25 and **0.324** for off-the-shelf SPLADE. That 29% improvement over BM25 translates to meaningfully better search results for real e-commerce queries. You can try the models directly from HuggingFace: [splade-ecommerce-esci](https://huggingface.co/Qdrant/splade-ecommerce-esci) (best in-domain) and [splade-ecommerce-multidomain](https://huggingface.co/Qdrant/splade-ecommerce-multidomain) (better generalization).
+The end result: a fine-tuned SPLADE model that achieves **nDCG@10 of 0.389** on Amazon ESCI, compared to **0.305** for BM25 and **0.326** for off-the-shelf SPLADE. That 28% improvement over BM25 translates to meaningfully better search results for real e-commerce queries. You can try the models directly from HuggingFace: [splade-ecommerce-esci](https://huggingface.co/Qdrant/splade-ecommerce-esci) (best in-domain) and [splade-ecommerce-multidomain](https://huggingface.co/Qdrant/splade-ecommerce-multidomain) (better generalization).
> **Note:** These metrics were measured on a subsample of 100k products and 10k queries where all relevant documents are included. They are not directly comparable to official Amazon ESCI benchmarks and should be treated as a comparative signal only.
diff --git a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md
index 7fbc171a6..8a87af1cb 100644
--- a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md
+++ b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md
@@ -1,7 +1,7 @@
---
title: "Fine-Tuning Sparse Embeddings for E-Commerce Search | Part 5: From Research to Product"
short_description: "One command to fine-tune SPLADE for your catalog. No ML pipeline assembly required."
-description: "Part 5 of the sparse embeddings series. We packaged the entire training pipeline from Parts 1-4 into an open-source CLI and web dashboard that fine-tunes SPLADE models for any product catalog in minutes."
+description: "Part 5 of a 5-part series on fine-tuning SPLADE sparse embeddings for e-commerce search. We packaged the entire training pipeline from Parts 1-4 into an open-source CLI and web dashboard that fine-tunes SPLADE models for any product catalog in minutes."
preview_dir: /articles_data/sparse-embeddings-ecommerce-part-5/preview
social_preview_image: /articles_data/sparse-embeddings-ecommerce-part-5/preview/social_preview.jpg
weight: 50
diff --git a/qdrant-landing/content/articles/sparse-vectors.md b/qdrant-landing/content/articles/sparse-vectors.md
index 28d8ee5a5..c2deb7a36 100644
--- a/qdrant-landing/content/articles/sparse-vectors.md
+++ b/qdrant-landing/content/articles/sparse-vectors.md
@@ -37,7 +37,7 @@ The numbers 331 and 14136 map to specific tokens in the vocabulary e.g. `['choco
The tokens aren't always words though, sometimes they can be sub-words: `['ch', 'ocolate']` too.
-They're pivotal in information retrieval, especially in ranking and search systems. BM25, a standard ranking function used by search engines like [Elasticsearch](https://www.elastic.co/blog/practical-bm25-part-2-the-bm25-algorithm-and-its-variables?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors), exemplifies this. BM25 calculates the relevance of documents to a given search query.
+They're pivotal in information retrieval, especially in ranking and search systems. [BM25](/documentation/search/text-search/full-text-search/#bm25), a standard ranking function natively available in Qdrant, exemplifies this. BM25 calculates the relevance of documents to a given search query.
BM25's capabilities are well-established, yet it has its limitations.
@@ -359,22 +359,20 @@ After setting up the collection and inserting sparse vectors, the next critical
```python
# Searching for similar documents
-result = client.search(
+result = client.query_points(
collection_name=COLLECTION_NAME,
- query_vector=models.NamedSparseVector(
- name="text",
- vector=models.SparseVector(
- indices=query_indices,
- values=query_values,
- ),
+ query=models.SparseVector(
+ indices=query_indices,
+ values=query_values,
),
+ using="text",
with_vectors=True,
-)
+).points
result
```
-In the above code, we execute a search against our collection using the prepared sparse vector query. The `client.search` method takes the collection name and the query vector as inputs. The query vector is constructed using the `models.NamedSparseVector`, which includes the indices and values derived from the query text. This is a crucial step in efficiently retrieving relevant documents.
+The `client.query_points` method takes the collection name and the sparse query vector as inputs. The `using` parameter selects the named vector.
```python
ScoredPoint(
diff --git a/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md b/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md
index dd3615af9..eda7a32bd 100644
--- a/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md
+++ b/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md
@@ -180,14 +180,10 @@ The created vectors might be easily put into Qdrant. For the sake of simplicity,
If you decided to describe each object with several [neural embeddings](https://qdrant.tech/articles/neural-search-tutorial/), then at each search operation you need to provide the vector name along with the [vector embedding](https://qdrant.tech/articles/what-are-embeddings/), so the engine knows which one to use. The interface of the search operation is pretty straightforward and requires an instance of NamedVector.
```python
-from qdrant_client.http.models import NamedVector
-
-text_results = client.search(
+text_results = client.query_points(
collection_name="ms-coco-2017",
- query_vector=NamedVector(
- name="text",
- vector=row["text_vector"],
- ),
+ query=row["text_vector"],
+ using="text",
limit=5,
with_vectors=False,
with_payload=True,
@@ -221,4 +217,4 @@ It is not surprising that a method used for creating neural encoding plays an im
- With Qdrant's new features, users can easily configure vector parameters, including size and distance functions, for each vector type, optimizing search results and user experience.
-If you’d like to check out some other examples, please check out our [full notebook](https://gist.github.com/kacperlukawski/961aaa7946f55110abfcd37fbe869b8f) presenting the search results and the whole pipeline implementation.
\ No newline at end of file
+If you’d like to check out some other examples, please check out our [full notebook](https://gist.github.com/kacperlukawski/961aaa7946f55110abfcd37fbe869b8f) presenting the search results and the whole pipeline implementation.
diff --git a/qdrant-landing/content/articles/tuning-qdrant-optimizer.md b/qdrant-landing/content/articles/tuning-qdrant-optimizer.md
new file mode 100644
index 000000000..1e889d38a
--- /dev/null
+++ b/qdrant-landing/content/articles/tuning-qdrant-optimizer.md
@@ -0,0 +1,190 @@
+---
+title: "Configure Qdrant's Optimizer for Predictable Search Latency"
+short_description: "Best practices for configuring Qdrant's optimizers, backed by search latency benchmarks."
+description: "Configuration guidance for Qdrant's indexing, merge, and vacuum optimizers, backed by search latency measurements across 13 configurations on a 1.76 million point collection."
+social_preview_image: /articles_data/tuning-qdrant-optimizer/preview/social_preview.jpg
+preview_dir: /articles_data/tuning-qdrant-optimizer/preview
+author: Clelia Bertelli
+date: 2026-08-25T10:00:00+02:00
+draft: false
+keywords:
+ - optimizer
+ - indexing
+ - vacuum
+ - read-write contention
+ - benchmark
+category: production-ops
+weight: 8
+---
+
+A bulk load finishes, and the collection looks ready: every point is in, and the upload call has returned. Then the first queries land, and search takes hundreds of milliseconds, sometimes several seconds at a stretch, while Qdrant's indexing, merge, and vacuum optimizers work through the backlog the upload left behind. How long that lasts, and what it costs each query, depends on settings most people never touch.
+
+Qdrant's [optimizer docs](/documentation/ops-optimization/optimizer/) and [read-write contention guide](/documentation/ops-optimization/read-write-contention/) already describe that trade-off qualitatively. To put numbers on it, we built a benchmark harness:
+
+- **Data.** 1.76 million Cohere-embedded MS MARCO passages.
+- **Hardware.** A single Qdrant node running Ubuntu 26.04 x86_64, with 32GB of RAM and 14 Intel CPU cores.
+- **Configurations.** 13 optimizer configurations, each measured both while the optimizer worked through its backlog and once it settled.
+- **Network.** The Qdrant instance ran on the same machine as the benchmark client, which reduces network latency: reproducing this benchmark against a cloud instance would likely show higher upload and search latencies.
+
+## How We Measured It
+
+Each run followed the same three stages, shown here as a timeline from the last point uploaded to a settled baseline:
+
+
+
+- **Upload.** All 1.76 million points go in with no search traffic running, so the collection is already full by the time we start measuring latency. Upload alone took anywhere from 70 to 316 seconds, depending on whether indexing was running concurrently with it.
+- **Draining.** Once the upload finishes, we search continuously (one query in flight at a time, no batching) while polling Qdrant's `/collections/{collection_name}/optimizations` endpoint every 2 seconds, until it reports nothing running and nothing queued.
+- **Steady.** With optimizers confirmed idle, we run five fixed passes over a separate set of 1,000 query vectors (5,000 searches) as the steady-state baseline.
+
+That closed-loop search pattern matters for reading the sample counts in the tables that follow.
+
+When a query takes 800 ms, only about 1.25 queries fit into a second of wall-clock time. A draining phase can run for over 10 minutes and still only collect a few hundred samples, while a steady phase with the same fixed 5,000-query workload finishes many times faster once nothing is competing with it. A small `n` during draining just reflects how slow search gets while optimizations are running, not missing data.
+
+A few notes on reading the numbers below:
+
+- All 13 collections lived on the same node for the whole test, so absolute latencies include that machine's baseline overhead. Read them as relative effects, not as a latency SLA for your own cluster.
+- The embeddings come from the `CohereLabs/msmarco-v2.1-embed-english-v3` dataset on Hugging Face, one segmented parquet file of MS MARCO v2.1 passages, 1024 dimensions per vector.
+- Latency figures throughout are p50 (median) and p95 (95th percentile) per-query times.
+
+## Continuous Indexing Pays Off (After a Recovery Window)
+
+We advise to leave continuous indexing on unless permanently slower search is acceptable. It costs a temporary recovery window right after ingestion, while the collection catches up on building the HNSW graph, but it results in a fully optimized index. Disabling indexing skips that window entirely, at the cost of brute-force scans for as long as indexing stays off.
+
+We validated this by comparing continuous indexing, where the HNSW graph builds while points are ingested, against indexing disabled by raising the HNSW build threshold:
+
+
+
+Continuous indexing needed a little over 11 minutes to clear the optimization backlog after the final point arrived. During that draining phase, search competed with optimization writes for the same resources: median latency was 780 ms, p95 reached 2.0 s, and a few queries took nearly 10 s.
+
+Once optimizers went idle, the picture flipped:
+
+- **Median search latency dropped 180x**, from 780 ms to 4.3 ms.
+- **p95 dropped 263x**, from 2.0 s to 7.6 ms.
+
+That gap is the cost of building a fully optimized HNSW graph while contending with live search, paid back in full once the graph is done.
+
+Indexing disabled skipped the draining phase entirely, since there were no indexing optimizations to complete. But steady-state search paid for that: with Qdrant falling back to brute-force scans, median latency was 256.6 ms, about 60 times higher than the continuously indexed collection after optimization.
+
+
+## `prevent_unoptimized` Buys Speed
+
+With continuous indexing, the experimental `prevent_unoptimized` flag (available since Qdrant 1.17.1) can reduce query latency under heavy write load. It works on the write path: once a growing segment's data crosses `indexing_threshold`, further points written to that segment become deferred points, durably stored but held back from search until the segment finishes optimizing. Already-indexed data stays fully searchable throughout.
+
+That's a different mechanism from the older `indexed_only` search parameter, which instead skips large unindexed segments at query time and can make points blink in and out of results as a segment crosses the threshold.
+
+In our analysis, enabling `prevent_unoptimized` dropped draining-phase p50 latency from 780 ms to 10.2 ms, a **76x improvement**, with p95 at 81.3 ms. Optimizations also completed faster, in about 9 minutes instead of 11, because search queries no longer competed with optimization work for the same resources.
+
+The left panel below shows the draining-phase latency drop; the right panel shows the shorter drain duration.
+
+
+
+**This speed comes with a trade-off.** Under heavy ingestion, freshly written points can sit as deferred for a while: durable, but invisible to search until their segment is optimized. Queries can return fewer results, or none for the most recent writes, until that backlog clears. Keep writes on `wait=false` while this is on. `wait=true` blocks until a point's deferred status clears, which can be slow enough to time out a client and head-of-line-block other writes.
+
+
+
+## Segment Size Trades Recovery Time for Query Speed
+
+Stick with Qdrant's default of one segment per CPU core unless a specific latency target pushes you to an extreme. A single segment gives the fastest steady-state search but takes over an hour to reach it. Capping segment size clears the backlog in under five minutes, at the cost of slower queries once everything settles.
+
+Fewer segments require more work from the `merge` optimizer, but result in a more compact HNSW index and faster searches. More segments reduce, or even remove, merge activity, but searches must traverse multiple segment-level indexes, which can increase latency.
+
+We tested four configurations:
+
+1. A single segment.
+2. Qdrant's default of one segment per CPU core.
+3. Four times the number of CPU cores.
+4. A smaller segment size of 100,000 KB (roughly 25,000 1024-dimensional full-precision vectors per segment).
+
+**Clearing the backlog.** The single-segment configuration was by far the slowest, taking just over one hour, because the `merge` optimizer had to consolidate all data into one segment on top of the indexing work. Capping segment size removes that merge cost entirely: the 100,000 KB configuration completed in 283.9 seconds, **12.7x faster**.
+
+
+
+**Search during draining.** A single segment performed poorly here: every query had to hit the same not-yet-fully-optimized segment, keeping latency consistently high. Qdrant's default fared better, since queries could increasingly land on already-optimized segments while only a shrinking share reached segments still being indexed. Adding more segments, either by raising the limit to four times the CPU count or by shrinking segment size, generally made draining latency slower and spikier than the default, trading it for a faster backlog cleanup.
+
+
+
+**Steady-state search.** Here the single segment won outright, with 3.2 ms median latency versus 4.1 ms for the default, 23.9 ms at four times the CPU-core count, and 17.3 ms with 100,000 KB segments.
+
+
+
+
+
+## Smoother Queries vs Shorter Wait
+
+Serialize optimizer threads if a smooth, predictable query latency during a bulk load matters more than how quickly the backlog clears. Leave Qdrant's default thread allocation in place if the opposite is true. Optimizers run on the same threads as your Qdrant instance, so limiting or increasing the number of threads they can use directly controls how fast they clear your collection's backlog and how much CPU capacity remains for search.
+
+Setting both `max_optimization_threads` and `max_indexing_threads` to 1 in our benchmark stretched the draining window to 3,244.1 seconds, **6.8 times longer than Qdrant's default settings**. In exchange, search latency during draining was lower and more predictable: with a limited CPU budget, the optimizers competed less with search operations, and p95 latency was capped at 373.7 ms, less than half of the default configuration's 820.4 ms. This matches the read/write contention trade-off the docs describe qualitatively.
+
+The chart below shows both effects together: draining latency on the left, drain duration on the right.
+
+
+
+
+
+## Vacuum: The Same Deletion, Two Opposite Outcomes
+
+Set `deleted_threshold` higher than the default if you can afford the extra disk space: letting soft-deleted points sit longer avoids triggering `vacuum` during active search traffic, which costs more than the storage it saves.
+
+Like many databases, Qdrant uses soft deletes: a `DELETE` request marks points as deleted, and queries skip them rather than immediately removing them from disk. This keeps delete operations fast, but leaves stale data in storage. Once the proportion of deleted points exceeds `deleted_threshold`, Qdrant's `vacuum` optimizer physically removes them, and like indexing, this background write activity can contend with searches.
+
+We compared a 20% threshold with a 50% threshold after deleting roughly 25% of the collection:
+
+- At 20%, vacuuming triggered, and **p95 search latency rose from 5.0 ms in steady state to 22.7 ms**.
+- At 50%, vacuuming did not run, and p95 latency fell from 5.4 ms to 4.3 ms, because fewer points remained searchable while soft-deleted points stayed on disk, avoiding expensive write operations.
+
+
+
+
+
+## Deferred Indexing Means Optimizing All at Once
+
+Don't defer indexing as a way to dodge read-write contention during upload unless you also enable `prevent_unoptimized` once you turn indexing back on. Flipping indexing on after the fact reopens the entire backlog at once, and without `prevent_unoptimized`, search competes with that backlog for as long as it takes to clear.
+
+We evaluated this by running the benchmark with indexing disabled, then reconfiguring the collection to activate indexing by lowering the indexing threshold, and measuring query latency over 5 rounds of 1,000 queries each. We ran this with both `prevent_unoptimized` set to `true` and to `false`.
+
+
+
+Without continuous indexing, collections lose the benefit of incremental index buildout during upload, so indexing takes longer and latency is higher once it resumes. From there, the two settings diverge sharply:
+
+- **`prevent_unoptimized: false`.** Optimizers never went idle, and **search latency climbed to an overall median of 2.7 s, with the tail reaching 12.1 s**. Search queries arrive continuously, repeatedly scanning the same growing backlog of unindexed points the optimizer hasn't caught up on, and compete with the optimizers for the same I/O and CPU resources.
+- **`prevent_unoptimized: true`.** Newly written points stayed deferred, durable but invisible to search, until their segment finished optimizing, so queries never scanned that backlog directly. Optimization progressed much faster, completing by the end of the first round of queries, and latency recovered fast: p95 came in at 621.7 ms, with a steady-state p95 of just 5.7 ms and median latency back down near 4.6 ms.
+
+As noted earlier, that recovery only applies if a temporary loss of recall for the most recent writes is acceptable in exchange for better query latency.
+
+## Takeaways
+
+- Try the experimental `prevent_unoptimized` flag before a bulk load if a short delay before new points become searchable is acceptable, and switch writes to `wait=false` first if your client defaults to `wait=true`. Confirm current behavior against your Qdrant version first, since this flag is still experimental and could change.
+- Watch `deferred_points` in the collection info while `prevent_unoptimized` is on. A nonzero count under load is normal; what matters is whether it drains.
+- Cap segment size for large loads if slower steady-state queries are an acceptable trade.
+- Do not assume serializing optimizer work is free.
+- Check `deleted_threshold` against your actual delete pattern, not just the default.
+- Budget for a real recovery window after a bulk load, not just the load itself.
+- Don't assume flipping indexing on later is gentler than running it from the start.
+
+## Caveats
+
+All 13 configurations ran against the same single-node Qdrant instance, one after another, so absolute latency numbers reflect that specific machine and shouldn't be read as general performance figures. The closed-loop search pattern also biases our draining-phase samples toward whatever finished fastest: a phase with severe contention produces fewer, noisier samples exactly when you would want more of them.
+
+That machine ran other work throughout, not just Qdrant, so a latency change inside a benchmark run isn't automatically proof of an optimizer effect. Two examples:
+
+- In the indexing-disabled run from the first section, we monitored the collection's memory via Qdrant's `/collections/{name}/memory` endpoint and found that search latency spiked five to six times at the median exactly when the OS reclaimed memory for other processes and Qdrant's own vector cache dropped with it.
+- In the first round of queries after reconfiguring indexing on with `prevent_unoptimized` disabled, some latency bursts had no such cache signal at all, more consistent with other processes competing for resources than with anything Qdrant was doing.
+
+We checked the optimizer status and memory status behind every result in this article before attributing it to indexing, merge, or vacuum specifically rather than to the machine. Both patterns are a caution for self-hosters who co-locate Qdrant with other workloads on the same box.
+
+This was a local, single-node setup. Qdrant Cloud results might differ, though the underlying trade-offs are expected to be directionally the same.
+
+## Adjacent Work
+
+- Qdrant's [optimizer docs](/documentation/ops-optimization/optimizer/) describe how the indexing, merge, and vacuum optimizers work and how to configure them.
+- The [read-write contention guide](/documentation/ops-optimization/read-write-contention/) explains why search and background optimization compete for the same CPU and I/O.
+- The [tutorial using `prevent_unoptimized`](/documentation/tutorials-operations/prevent-unoptimized-usage/) explains how the flag affects search latency and shows how to measure it.
+- The full benchmarks, including harness, scripts, and results, are available on GitHub at [qdrant-labs/optimizers-in-action](https://github.com/qdrant-labs/optimizers-in-action).
diff --git a/qdrant-landing/content/articles/vector-search-production.md b/qdrant-landing/content/articles/vector-search-production.md
index b411e7d30..188682f9e 100644
--- a/qdrant-landing/content/articles/vector-search-production.md
+++ b/qdrant-landing/content/articles/vector-search-production.md
@@ -243,7 +243,7 @@ It depends. If you're just starting out - we have prepared a tool on our website
A three-node setup provides a baseline for fault tolerance: if one node goes offline, the remaining two can continue serving queries and maintain a quorum for data consistency. This guards against hardware failures, rolling updates, and network disruptions. Fewer than three nodes leaves you vulnerable to single-point failures that can knock your entire cluster offline.
-> [**We follow the Raft Protocol**](https://qdrant.tech/documentation/distributed_deployment/#raft), so check out the docs and learn why this is important.
+> [**We follow the Raft Protocol**](https://qdrant.tech/documentation/scaling/horizontal-scaling/#raft-consensus), so check out the docs and learn why this is important.
✅ **Set a replication factor of at least 2** to tolerate node failure without losing availability.
@@ -275,7 +275,7 @@ Development and staging environments often run experimental builds, tests, or si
> It's quite possible that the user has multiple shards on one node, which end up handling most traffic while other nodes remain underutilized.
-In this case, you should [**choose the right number of shards**](https://qdrant.tech/documentation/distributed_deployment/#sharding) based on your node count and expected RPS.
+In this case, you should [**choose the right number of shards**](https://qdrant.tech/documentation/scaling/distributed_deployment/#sharding) based on your node count and expected RPS.
You need to implement a shard strategy that aligns with real usage patterns. First, distribute your shards across all available nodes. This will help balance the load more effectively. After redistributing the shards, run performance tests to see how it affects your system. Then add replicas and test again to see how that changes performance.
@@ -286,14 +286,14 @@ Proper sharding considers data distribution and query patterns. By default, shar
||
|-|
-|**Read More:** [**Sharding Documentation**](https://qdrant.tech/documentation/distributed_deployment/#sharding)|
+|**Read More:** [**Sharding Documentation**](https://qdrant.tech/documentation/scaling/distributed_deployment/#sharding)|
### Manage Your Costs by Scaling Up or Down

Some teams scale up for daytime surges, then scale down overnight to save resources. If you do this, ensure data is sharded and replicated appropriately, so that scaling up and down won't result in service degradation.
-If using Qdrant Cloud you could also do this using the [**Replication Factor**](https://qdrant.tech/documentation/distributed_deployment/#replication-factor), though it may be considered a bit of a hack.
+If using Qdrant Cloud you could also do this using the [**Replication Factor**](https://qdrant.tech/documentation/scaling/distributed_deployment/#replication-factor), though it may be considered a bit of a hack.
> If you have 3 nodes with just 1 shard, and replication factor 6. It will create 3 replicas (one on each node) of that shard, because it can't host more. If you add 3 more nodes at peak times, it'll automatically replicate that shard 3 more times in an attempt to match the factor of 6.
@@ -309,7 +309,7 @@ If new nodes remain empty after joining, you waste resources. If departing nodes
||
|-|
-|**Read More:** [**Distributed Deployment Documentation**](https://qdrant.tech/documentation/distributed_deployment/)|
+|**Read More:** [**Distributed Deployment Documentation**](https://qdrant.tech/documentation/scaling/distributed_deployment/)|
|**Read More:** [**Resharding**](https://qdrant.tech/documentation/cloud/cluster-scaling/#resharding)|
### How to Predict and Test Cluster Performance
@@ -332,7 +332,7 @@ Remember, cold-starts and query behaviour are dataset dependent, which is why yo
||
|-|
-|**Read More:** [Distributed Deployment Documentation](https://qdrant.tech/documentation/distributed_deployment/)
+|**Read More:** [Distributed Deployment Documentation](https://qdrant.tech/documentation/scaling/distributed_deployment/)
### How to Design Your Systems to Protect Against Failure
diff --git a/qdrant-landing/content/articles/vector-search-resource-optimization.md b/qdrant-landing/content/articles/vector-search-resource-optimization.md
index 21f326397..bbe770dcc 100644
--- a/qdrant-landing/content/articles/vector-search-resource-optimization.md
+++ b/qdrant-landing/content/articles/vector-search-resource-optimization.md
@@ -325,7 +325,7 @@ Here’s how to choose the shard_number:
| **Plan for Scalability** | Start with at least **2 shards per node** to allow room for future growth. |
| **Future-Proofing** | Starting with around **12 shards** is a good rule of thumb. This setup allows your system to scale seamlessly from 1 to 12 nodes without requiring re-sharding. |
-Learn more about [**Sharding in Distributed Deployment**](/documentation/distributed_deployment/)
+Learn more about [**Sharding in Distributed Deployment**](/documentation/scaling/distributed_deployment/)
---
@@ -342,9 +342,9 @@ The filterable vector index is Qdrant's solves pre and post-filtering problems b
**Example:**
```python
-results = client.search(
+results = client.query_points(
collection_name="my_collection",
- query_vector=[0.1, 0.2, 0.3],
+ query=[0.1, 0.2, 0.3],
query_filter=models.Filter(must=[
models.FieldCondition(
key="category",
@@ -456,7 +456,7 @@ The rescoring process maps the quantized vectors to their corresponding original
```python
client.query_points(
collection_name="my_collection",
- query_vector=[0.22, -0.01, -0.98, 0.37],
+ query=[0.22, -0.01, -0.98, 0.37],
search_params=models.SearchParams(
quantization=models.QuantizationSearchParams(
rescore=True, # Enables rescoring with original vectors
@@ -601,4 +601,4 @@ _________________________________________________________________________
Want to download a printer-friendly version of this guide? [**Download it now.**](https://try.qdrant.tech/resource-optimization-guide).
-[](https://try.qdrant.tech/resource-optimization-guide)
\ No newline at end of file
+[](https://try.qdrant.tech/resource-optimization-guide)
diff --git a/qdrant-landing/content/articles/what-is-a-vector-database.md b/qdrant-landing/content/articles/what-is-a-vector-database.md
index ce75228fe..866c31cdb 100644
--- a/qdrant-landing/content/articles/what-is-a-vector-database.md
+++ b/qdrant-landing/content/articles/what-is-a-vector-database.md
@@ -1,14 +1,14 @@
---
title: "What is a Vector Database?"
draft: false
-short_description: What is a Vector Database? Use Cases & Examples | Qdrant
-description: Discover what a vector database is, its core functionalities, and real-world applications.
+short_description: What Is a Vector Database? Concepts, Architecture & Use Cases | Qdrant
+description: Discover how vector databases power semantic search by understanding the meaning behind your data. This guide breaks down how they work and why they've become the backbone of modern AI search.
preview_dir: /articles_data/what-is-a-vector-database/preview
weight: 30
social_preview_image: /articles_data/what-is-a-vector-database/preview/social_preview.png
date: 2024-10-09T09:29:33-03:00
aliases: [ /blog/what-is-a-vector-database/ ]
-author: Sabrina Aquino
+author: Sabrina Aquino & Chadha Sridi
featured: true
tags:
- vector-search
@@ -19,19 +19,15 @@ category: core-concepts
## An Introduction to Vector Databases
-
-
Most of the millions of terabytes of data we generate each day is **unstructured**. Think of the meal photos you snap, the PDFs shared at work, or the podcasts you save but may never listen to. None of it fits neatly into rows and columns.
Unstructured data lacks a strict format or schema, making it challenging for conventional databases to manage. Yet, this unstructured data holds immense potential for **AI**, **machine learning**, and **modern search engines**.
-> A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized system designed to efficiently handle high-dimensional vector data. It excels at indexing, querying, and retrieving this data, enabling advanced analysis and similarity searches that traditional databases cannot easily perform.
-
### The Challenge with Traditional Databases
Traditional [OLTP](https://www.ibm.com/topics/oltp) and [OLAP](https://www.ibm.com/topics/olap) databases have been the backbone of data storage for decades. They are great at managing structured data with well-defined schemas, like `name`, `address`, `phone number`, and `purchase history`.
-
+
But when data can't be easily categorized, like the content inside a PDF file, things start to get complicated.
@@ -39,13 +35,15 @@ You can always store the PDF file as raw data, perhaps with some metadata attach
Also, this applies to more than just PDF documents. Think about the vast amounts of text, audio, and image data you generate every day. If a database can’t grasp the **meaning** of this data, how can you search for or find relationships within the data?
-
+
-Vector databases allow you to understand the **context** or **conceptual similarity** of unstructured data by representing them as vectors, enabling advanced analysis and retrieval based on data similarity.
+This is where vector databases come in. They index and query unstructured data as **vectors** that capture patterns and relationships, enabling applications to search and retrieve information based on meaning rather than exact matches.
-## When to Use a Vector Database
+> A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized system designed to efficiently handle high-dimensional vector data. It excels at indexing, querying, and retrieving this data, enabling advanced analysis and similarity searches that traditional databases cannot easily perform.
-Not sure if you should use a vector database or a traditional database? This chart may help.
+## Traditional Databases vs Vector Databases
+
+Traditional and vector databases aren't rivals; they solve different problems. The table below shows how they compare across data structure, query method, and typical use cases.
| **Feature** | **OLTP Database** | **OLAP Database** | **Vector Database** |
|---------------------|--------------------------------------|--------------------------------------------|--------------------------------------------|
@@ -56,52 +54,95 @@ Not sure if you should use a vector database or a traditional database? This cha
| **Performance** | Optimized for high-volume transactions | Optimized for complex analytical queries | Optimized for unstructured data retrieval |
| **Use Cases** | Inventory, order processing, CRM | Business intelligence, data warehousing | Similarity search, recommendations, RAG, anomaly detection, etc. |
-
## What Is a Vector?
-
-
When a machine needs to process unstructured data - an image, a piece of text, or an audio file, it first has to translate that data into a format it can work with: **vectors**.
-> A **vector** is a numerical representation of data that can capture the **context** and **semantics** of data.
-
-When you deal with unstructured data, traditional databases struggle to understand its meaning. However, a vector can translate that data into something a machine can process. For example, a vector generated from text can represent relationships and meaning between words, making it possible for a machine to compare and understand their context.
-
-There are three key elements that define a vector in a vector database: the **ID**, the **dimensions**, and the **payload**. These components work together to represent a vector effectively within the system. Together, they form a **point**, which is the core unit of data stored and retrieved in a vector database.
-
-
-
-Each one of these parts plays an important role in how vectors are stored, retrieved, and interpreted. Let's see how.
-
-### 1. The ID: Your Vector’s Unique Identifier
-
-Just like in a relational database, each vector in a vector database gets a unique ID. Think of it as your vector’s name tag, a **primary key** that ensures the vector can be easily found later. When a vector is added to the database, the ID is created automatically.
-
-While the ID itself doesn't play a part in the similarity search (which operates on the vector's numerical data), it is essential for associating the vector with its corresponding "real-world" data, whether that’s a document, an image, or a sound file.
-
-After a search is performed and similar vectors are found, their IDs are returned. These can then be used to **fetch additional details or metadata** tied to the result.
-
-### 2. The Dimensions: The Core Representation of the Data
-
-At the core of every vector is a set of numbers, which together form a representation of the data in a **multi-dimensional** space.
-
-#### From Text to Vectors: How Does It Work?
+> A **vector** is a numerical representation of data in a **multi-dimensional space** that captures the **context** and **semantics** of data.
These numbers are generated by **embedding models**, such as deep learning algorithms, and capture the essential patterns or relationships within the data. That's why the term **embedding** is often used interchangeably with vector when referring to the output of these models.
To represent textual data, for example, an embedding will encapsulate the nuances of language, such as semantics and context within its dimensions.
-
+
For that reason, when comparing two similar sentences, their embeddings will turn out to be very similar, because they have similar **linguistic elements**.
-
+
That’s the beauty of embeddings. The complexity of the data is distilled into something that can be compared across a multi-dimensional space.
-### 3. The Payload: Adding Context with Metadata
+## The Vector Search Workflow
-Sometimes you're going to need more than just numbers to fully understand or refine a search. While the dimensions capture the essence of the data, the payload holds **metadata** for structured information.
+Before diving into the individual concepts, it helps to see the full picture. Regardless of the use case, the underlying vector search workflow consists of two complementary paths: the **ingestion path**, where data is processed and stored as vectors, and the **query path**, where user queries are matched against the stored vectors.
+
+### Ingestion Path
+
+1. **Generate embeddings:** An embedding model converts raw data into vectors that capture its semantic meaning.
+
+2. **Store and index the vectors:** The vectors, along with any associated metadata, are stored in the vector database, which builds an index for efficient similarity search.
+
+### Query Path
+
+3. **Embed the query:** When a user submits a search query, the same embedding model converts it into a vector.
+
+4. **Search for similar vectors:** The vector database compares the query vector against the indexed vectors and returns the closest matches, ranked by similarity.
+
+5. **Power your application:** The retrieved results can then be used in applications such as semantic search, recommendation systems, anomaly detection, and Retrieval-Augmented Generation (RAG).
+
+
+
+## Common Applications of Vector Search
+
+Vector search enables a wide range of applications by allowing systems to retrieve information based on meaning rather than exact matches. Some of the most common use cases include:
+
+| **Use Case** | **How It Works** | **Examples** |
+|-----------------------------------|------------------------------------------------------------------------------------------------------|-----------------------------------------------------------|
+| **Similarity Search** | Finds similar data points using vector distances | Find similar product images, retrieve documents based on themes, discover related topics |
+| **Anomaly Detection** | Identifies outliers based on deviations in vector space | Detect unusual user behavior in banking, spot irregular patterns |
+| **Recommendation Systems** | Uses vector embeddings to learn and model user preferences | Personalized movie or music recommendations, e-commerce product suggestions |
+| **RAG (Retrieval-Augmented Generation)** | Combines vector search with large language models (LLMs) for contextually relevant answers | Customer support, auto-generate summaries of documents, research reports |
+| **Multimodal Search** | Search across different types of data like text, images, and audio in a single query. | Search for products with a description and image, retrieve images based on audio or text |
+| **Voice & Audio Recognition** | Uses vector representations to recognize and retrieve audio content | Speech-to-text transcription, voice-controlled smart devices, identify and categorize sounds |
+| **Knowledge Graph Augmentation** | Links unstructured data to concepts in knowledge graphs using vectors | Link research papers to related studies, connect customer reviews to product features, organize patents by innovation trends|
+
+## The Architecture of a Vector Database
+
+> A quick note on naming. You'll almost always hear these systems called vector databases, but the term is a little misleading. Traditional databases like Postgres or MySQL are built on ACID principles: transactions, strong consistency, and atomicity. Most "vector databases" aren't databases in that sense. They're really **vector search engines**, designed for horizontal scalability, low-latency queries, and high availability. Those priorities lead to different architectural decisions that are not reproducible in general-purpose databases. The label vector database stuck for marketing reasons, so we use it throughout this article, but vector search engine is the more accurate description.
+
+A vector database is made of multiple different entities and relations. Let's understand a bit of what's happening here:
+
+
+### Points
+[Points](https://qdrant.tech/documentation/manage-data/points/) are the core units of data stored and retrieved. They are the central entity that Qdrant operates with.
+
+There are three key elements that form a point: the **ID**, one or more **vector(s)**, and the **payload**.
+
+
+Each one of these parts plays an important role in how a point is stored, retrieved, and interpreted. Let's see how.
+
+#### 1. The ID: The Point's Unique Identifier
+
+Just like in a relational database, each point in a vector database gets a unique ID that ensures it can be easily found later. When a vector is added to the database, the ID is created automatically.
+
+While the ID itself doesn't play a part in the similarity search (which operates on the vector's numerical data), it is essential for associating the vector with its corresponding "real-world" data, whether that’s a document, an image, or a sound file.
+
+After a search is performed and similar vectors are found, their IDs are returned. These can then be used to **fetch additional details or metadata** tied to the result.
+
+#### 2. The Vector(s): One or More Representations of the Data
+
+The vector is the numerical representation that powers similarity search (see [What Is a Vector?](#what-is-a-vector)).
+
+It is possible to attach more than one vector to a single point. In Qdrant we call these **named vectors**.
+
+With named vectors, one point can hold several vectors, each with its own name, dimensionality, and distance metric.
+
+This lets you store multiple representations of the same object side by side, for example a **dense** and a **sparse** vector for the same piece of text, or separate embeddings for an image and its caption. At search time you can combine these representations to power [hybrid search](#hybrid-search).
+
+
+#### 3. The Payload: Adding Context with Metadata
+
+Sometimes you're going to need more than just numbers to fully understand or refine a search. While the vectors capture the essence of the data, the payload holds **metadata** for structured information.
It could be textual data like descriptions, tags, categories, or it could be numerical values like dates or prices. This extra information is vital when you want to filter or rank search results based on criteria that aren’t directly encoded in the vector.
@@ -111,26 +152,22 @@ For example, if you’re searching for a picture of a dog, the vector helps the
-The payload can help you narrow down those results by ignoring vectors that don't match your query vector filtering criteria. If you want the full picture of how filtering works in Qdrant, check out our [Complete Guide to Filtering.](https://qdrant.tech/articles/vector-search-filtering/)
-
-## The Architecture of a Vector Database
-
-A vector database is made of multiple different entities and relations. Let's understand a bit of what's happening here:
-
+The payload can help you narrow down those results by ignoring vectors that don't match your filtering criteria. If you want the full picture of how filtering works in Qdrant, check out our [Complete Guide to Filtering.](https://qdrant.tech/articles/vector-search-filtering/)
### Collections
-A [collection](https://qdrant.tech/documentation/manage-data/collections/) is essentially a group of **vectors** (or “[points](https://qdrant.tech/documentation/manage-data/points/)”) that are logically grouped together **based on similarity or a specific task**. Every vector within a collection shares the same dimensionality and can be compared using a single metric. Avoid creating multiple collections unless necessary; instead, consider techniques like **sharding** for scaling across nodes or **multitenancy** for handling different use cases within the same infrastructure.
+A [collection](https://qdrant.tech/documentation/manage-data/collections/) is essentially a set of points that you can search over together. Within a collection, vectors of the same name must share the same dimensionality and be compared with a single distance metric.
+Avoid creating multiple collections unless necessary; instead, consider techniques like **sharding** for scaling across nodes or **multitenancy** for handling different use cases within the same infrastructure.
### Distance Metrics
-These metrics defines how similarity between vectors is calculated. The choice of distance metric is made when creating a collection and the right choice depends on the type of data you’re working with and how the vectors were created. Here are the three most common distance metrics:
+These metrics define how similarity between vectors is calculated. The choice of distance metric is made when creating a collection and the right choice depends on the type of data you’re working with and how the vectors were created. Here are the three most common distance metrics:
- **Euclidean Distance:** The straight-line path. It’s like measuring the physical distance between two points in space. Pick this one when the actual distance (like spatial data) matters.
- **Cosine Similarity:** This one is about the angle, not the length. It measures how two vectors point in the same direction, so it works well for text or documents when you care more about meaning than magnitude. For example, if two things are *similar*, *opposite*, or *unrelated*:
-
+
- **Dot Product:** This looks at how much two vectors align. It’s popular in recommendation systems where you're interested in how much two things “agree” with each other.
@@ -157,27 +194,26 @@ For other configurations like `hnsw_config.on_disk` or `memmap_threshold`, see t
### SDKs
-Qdrant offers a range of SDKs. You can use the programming language you're most comfortable with, whether you're coding in [Python](https://github.com/qdrant/qdrant-client), [Go](https://github.com/qdrant/go-client), [Rust](https://github.com/qdrant/rust-client), [Javascript/Typescript](https://github.com/qdrant/qdrant-js), [C#](https://github.com/qdrant/qdrant-dotnet) or [Java](https://github.com/qdrant/java-client).
+Qdrant offers a range of SDKs. You can use the programming language you're most comfortable with, whether you're coding in [Python](https://github.com/qdrant/qdrant-client), [Go](https://github.com/qdrant/go-client), [Rust](https://github.com/qdrant/rust-client), [JavaScript/TypeScript](https://github.com/qdrant/qdrant-js), [C#](https://github.com/qdrant/qdrant-dotnet) or [Java](https://github.com/qdrant/java-client).
## The Core Functionalities of Vector Databases
-
-
When you think of a traditional database, the operations are familiar: you **create**, **read**, **update**, and **delete** records. These are the fundamentals. And guess what? In many ways, vector databases work the same way, but the operations are translated for the complexity of vectors.
-### 1. Indexing: HNSW Index and Sending Data to Qdrant
+### 1. Indexing
Indexing your vectors is like creating an entry in a traditional database. But for vector databases, this step is very important. Vectors need to be indexed in a way that makes them easy to search later on.
+#### 1.1 HNSW Indexing
**HNSW** (Hierarchical Navigable Small World) is a powerful indexing algorithm that most vector databases rely on to organize vectors for fast and efficient search.
It builds a multi-layered graph, where each vector is a node and connections represent similarity. The higher layers connect broadly similar vectors, while lower layers link vectors that are closely related, making searches progressively more refined as they go deeper.
-
+
When you run a search, HNSW starts at the top, quickly narrowing down the search by hopping between layers. It focuses only on relevant vectors as it goes deeper, refining the search with each step.
-### 1.1 Payload Indexing
+#### 1.2 Payload Indexing
In Qdrant, indexing is modular. You can configure indexes for **both vectors and payloads independently**. The payload index is responsible for optimizing filtering based on metadata. Each payload index is built for a specific field and allows you to quickly filter vectors based on specific conditions.
@@ -191,7 +227,7 @@ You need to build the payload index for **each field** you'd like to search. The
Similarity search allows you to search by **meaning**. This way you can do searches such as similar songs that evoke the same mood, finding images that match your artistic vision, or even exploring emotional patterns in text.
-
+
The way it works is, when the user queries the database, this query is also converted into a vector. The algorithm quickly identifies the area of the graph likely to contain vectors closest to the **query vector**.
@@ -201,13 +237,13 @@ The search then moves down progressively narrowing down to more closely related
Here's a high-level overview of this process:
-
+
-### 3. Updating Vectors: Real-Time and Bulk Adjustments
+### 3. Updating Points: Real-Time and Bulk Adjustments
-Data isn't static, and neither are vectors. Keeping your vectors up to date is crucial for maintaining relevance in your searches.
+Data isn't static, and neither are vectors. Keeping your points up to date is crucial for maintaining relevance in your searches.
-Vector updates don’t always need to happen instantly, but when they do, Qdrant handles real-time modifications efficiently with a simple API call:
+Point updates don’t always need to happen instantly, but when they do, Qdrant handles real-time modifications efficiently with a simple API call:
```python
client.upsert(
@@ -216,7 +252,7 @@ client.upsert(
)
```
-For large-scale changes, like re-indexing vectors after a model update, batch updating allows you to update multiple vectors in one operation without impacting search performance:
+For large-scale changes, like re-indexing vectors after a model update, batch updating allows you to update multiple points in one operation without impacting search performance:
```python
batch_of_updates = [
@@ -231,11 +267,11 @@ client.upsert(
)
```
-### 4. Deleting Vectors: Managing Outdated and Duplicate Data
+### 4. Deleting Points: Managing Outdated and Duplicate Data
-Efficient vector management is key to keeping your searches accurate and your database lean. Deleting vectors that represent outdated or irrelevant data, such as expired products, old news articles, or archived profiles, helps maintain both performance and relevance.
+Efficient vector management is key to keeping your searches accurate and your database lean. Deleting points that represent outdated or irrelevant data, such as expired products, old news articles, or archived profiles, helps maintain both performance and relevance.
-In Qdrant, removing vectors is straightforward, requiring only the vector IDs to be specified:
+In Qdrant, removing points is straightforward, requiring only the point IDs to be specified:
```python
client.delete(
@@ -245,29 +281,34 @@ client.delete(
```
You can use deletion to remove outdated data, clean up duplicates, and manage the lifecycle of vectors by automatically deleting them after a set period to keep your dataset relevant and focused.
-## Dense vs. Sparse Vectors
-
+## Hybrid Search
-Now that you understand what vectors are and how they are created, let's learn more about the two possible types of vectors you can use: **dense** or **sparse**. The main difference between the two are:
+Sometimes context alone isn’t enough. Sometimes you need precision, too. Dense vectors (embeddings we've talked about so far) are fantastic when you need to retrieve results based on the context or meaning behind the data. Sparse vectors are useful when you also need **keyword or specific attribute matching**.
-### 1. Dense Vectors
+> With hybrid search you don’t have to choose one over the other and use both to get searches that are more **relevant** and **filtered**.
+
+### Dense vs. Sparse Vectors
+
+Dense and sparse vectors represent data in fundamentally different ways, and each excels at a different type of retrieval. Understanding their strengths and limitations explains why combining them can produce better search results than relying on either approach alone.
+
+#### 1. Dense Vectors
Dense vectors are, quite literally, dense with information. Every element in the vector contributes to the **semantic meaning**, **relationships** and **nuances** of the data. A dense vector representation of this sentence might look like this:
-
+
Each number holds weight. Together, they convey the overall meaning of the sentence, and are better for identifying contextually similar items, even if the words don’t match exactly.
-### 2. Sparse Vectors
+#### 2. Sparse Vectors
-Sparse vectors operate differently. They focus only on the essentials. In most sparse vectors, a large number of elements are zeros. When a feature or token is present, it’s marked—otherwise, zero.
+Sparse vectors operate differently. They focus only on the essentials. In most sparse vectors, a large number of elements are zeros. When a feature or token is present, it’s marked; otherwise, zero.
In the image, you can see a sentence, *“I love Vector Similarity,”* broken down into tokens like *“i,” “love,” “vector”* through tokenization. Each token is assigned a unique `ID` from a large vocabulary. For example, *“i”* becomes `193`, and *“vector”* becomes `15012`.
-
+
-Sparse vectors, are used for **exact matching** and specific token-based identification. The values on the right, such as `193: 0.04` and `9182: 0.12`, are the scores or weights for each token, showing how relevant or important each token is in the context. The final result is a sparse vector:
+Sparse vectors are used for **exact matching** and specific token-based identification. The values on the right, such as `193: 0.04` and `9182: 0.12`, are the scores or weights for each token, showing how relevant or important each token is in the context. The final result is a sparse vector:
```json
{
@@ -281,81 +322,77 @@ Sparse vectors, are used for **exact matching** and specific token-based identif
Everything else in the vector space is assumed to be zero.
-Sparse vectors are ideal for tasks like **keyword search** or **metadata filtering**, where you need to check for the presence of specific tokens without needing to capture the full meaning or context. They suited for exact matches within the **data itself**, rather than relying on external metadata, which is handled by payload filtering.
+Sparse vectors are ideal for tasks like **keyword search** where you need to check for the presence of specific tokens without needing to capture the full meaning or context. They are suited for exact matches within the **data itself**.
-## Benefits of Hybrid Search
+> Unlike dense vectors, which are indexed using HNSW, sparse vectors use a different indexing structure called an **inverted index**. An inverted index maps each non-zero dimension (token ID) to the vectors that contain it, along with the corresponding weights. At query time, only vectors sharing non-zero dimensions with the query are considered and scored using **dot product** based on their overlapping terms.
-
+### Fusion
-Sometimes context alone isn’t enough. Sometimes you need precision, too. Dense vectors are fantastic when you need to retrieve results based on the context or meaning behind the data. Sparse vectors are useful when you also need **keyword or specific attribute matching**.
+Qdrant uses **normalization** and **fusion** techniques to blend results from multiple search methods, such as dense and sparse. One common approach is **Reciprocal Rank Fusion (RRF)**, where results from different methods are merged, giving higher importance to items ranked highly by both methods. This ensures that the best candidates, whether identified through dense or sparse vectors, appear at the top of the results.
-> With hybrid search you don’t have to choose one over the other and use both to get searches that are more **relevant** and **filtered**.
-
-To achieve this balance, Qdrant uses **normalization** and **fusion** techniques to blend results from multiple search methods. One common approach is **Reciprocal Rank Fusion (RRF)**, where results from different methods are merged, giving higher importance to items ranked highly by both methods. This ensures that the best candidates, whether identified through dense or sparse vectors, appear at the top of the results.
-
-Qdrant combines dense and sparse vector results through a process of **normalization** and **fusion**.
-
-
+
### How to Use Hybrid Search in Qdrant
Qdrant makes it easy to implement hybrid search through its Query API. Here’s how you can make it happen in your own project:
-
+
-**Example Hybrid Query:** Let’s say a researcher is looking for papers on NLP, but the paper must specifically mention "transformers" in the content:
+**Example Hybrid Query:**
+
+```python
+client.query_points(
+ collection_name="{collection_name}",
+ prefetch=[
+ models.Prefetch(
+ query=models.SparseVector(indices=[1, 42], values=[0.22, 0.8]),
+ using="sparse",
+ limit=20,
+ ),
+ models.Prefetch(
+ query=[0.01, 0.45, 0.67], # <-- dense vector
+ using="dense",
+ limit=20,
+ ),
+ ],
+ query=models.RrfQuery(rrf=models.Rrf()),
+)
-```json
-search_query = {
- "vector": query_vector, # Dense vector for semantic search
- "filter": { # Filtering for specific terms
- "must": [
- {"key": "text", "match": "transformers"} # Exact keyword match in the paper
- ]
- }
-}
```
-
-In this query the dense vector search finds papers related to the broad topic of NLP and the sparse vector filtering ensures that the papers specifically mention “transformers”.
-
This is just a simple example and there's so much more you can do with it. See our complete [article on Hybrid Search](https://qdrant.tech/articles/hybrid-search/) guide to see what's happening behind the scenes and all the possibilities when building a hybrid search system.
-## Quantization: Get 40x Faster Results
+## Quantization
-
+As your vector dataset grows larger, so do the memory and compute demands of searching through it. Quantization compresses vectors into a more compact form that is cheaper to store and faster to compare, trading a small amount of precision for large gains in efficiency.
-As your vector dataset grows larger, so do the computational demands of searching through it.
+Qdrant 1.18 ships [**TurboQuant**](https://qdrant.tech/articles/turboquant-quantization/), a rotation-based vector quantization method from Google Research, with extensions that make it work on real production embeddings.
-Quantized vectors are much smaller and easier to compare. With methods like [**Binary Quantization**](https://qdrant.tech/articles/binary-quantization/), you can see **search speeds improve by up to 40x while memory usage decreases by 32x**. Improvements that can be decisive when dealing with large datasets or needing low-latency results.
+Instead of storing each dimension of your vectors at full 32-bit precision, quantization compresses it down to just a few bits. TurboQuant's 4-bit variant gives you 8x compression while matching Scalar Quantization's recall at half the memory.
+When you need to squeeze memory further, the 2-bit and 1-bit variants push the compression rate to 16x and 32x while still beating Binary Quantization at the same storage budget.
-It works by converting high-dimensional vectors, which typically use `4 bytes` per dimension, into binary representations, using just `1 bit` per dimension. Values above zero become "1", and everything else becomes "0".
+
-
-
-Quantization reduces data precision, and yes, this does lead to some loss of accuracy. However, for binary quantization, **OpenAI embeddings** achieves this performance improvement at a cost of only 5% of accuracy. If you apply techniques like **oversampling** and **rescoring**, this loss can be brought down even further.
-
-However, binary quantization isn’t the only available option. Techniques like [**Scalar Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#scalar-quantization) and [**Product Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#product-quantization) are also popular alternatives when optimizing vector compression.
-
-You can set up your chosen quantization method using the `quantization_config` parameter when creating a new collection:
+You can set up quantization using the `quantization_config` parameter when creating a new collection:
```python
client.create_collection(
collection_name="{collection_name}",
vectors_config=models.VectorParams(
- size=1536,
+ size=1536,
distance=models.Distance.COSINE
),
- # Choose your preferred quantization method
- quantization_config=models.BinaryQuantization(
- binary=models.BinaryQuantizationConfig(
- always_ram=True, # Store the quantized vectors in RAM for faster access
+ # Enable TurboQuant (4-bit by default)
+ quantization_config=models.TurboQuantization(
+ turbo=models.TurboQuantQuantizationConfig(
+ always_ram=True, # Keep quantized vectors in RAM for faster access
),
),
)
```
-You can store original vectors on disk within the `vectors_config` by setting `on_disk=True` to save RAM space, while keeping quantized vectors in RAM for faster access
-We recommend checking out our [Vector Quantization guide](https://qdrant.tech/articles/what-is-vector-quantization/) for a full breakdown of methods and tips on **optimizing performance** for your specific use case.
+You can store the original vectors on disk within `vectors_config` by setting `on_disk=True` to save RAM, while keeping the quantized vectors in RAM for faster access.
+
+TurboQuant isn't the only option. [**Binary Quantization**](https://qdrant.tech/articles/binary-quantization/) is the most aggressive choice for speed, converting each dimension to a single bit so that search speeds can improve by up to 40x with memory reduced by 32x, at an accuracy cost that oversampling and rescoring can recover. [**Scalar Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#scalar-quantization) and [**Product Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#product-quantization) round out the alternatives. Check out our [Vector Quantization guide](https://qdrant.tech/articles/what-is-vector-quantization/) for a full breakdown and tips on **optimizing performance** for your use case.
## Distributed Deployment
@@ -363,11 +400,11 @@ When thinking about scaling, the key factors to consider are **fault tolerance**
### Sharding: Distributing Data Across Nodes
-In a distributed Qdrant cluster, data is split into smaller units called **shards**, which are distributed across different nodes. which helps balance the load and ensures that queries can be processed in parallel.
+In a distributed Qdrant cluster, data is split into smaller units called **shards**, which are distributed across different nodes. This helps balance the load and ensures that queries can be processed in parallel.
-Each collection—a group of related data points—can be split into non-overlapping subsets, which are then managed by different nodes.
+Each collection can be split into non-overlapping subsets, which are then managed by different nodes.
-
+
**Raft Consensus** ensures that all the nodes stay in sync and have a consistent view of the data. Each node knows where every shard is, and Raft ensures that all nodes are in sync. If one node fails, the others know where the missing data is located and can take over.
@@ -386,9 +423,9 @@ There are two main types of sharding:
1. **Automatic Sharding:** Points (vectors) are automatically distributed across shards using consistent hashing. Each shard contains non-overlapping subsets of the data.
2. **User-defined Sharding:** Specify how points are distributed, enabling more control over your data organization, especially for use cases like **multitenancy**, where each tenant (a user, client, or organization) has their own isolated data.
-Each shard is divided into **segments**. They are a smaller storage unit within a shard, storing a subset of vectors and their associated payloads (metadata). When a query is executed, it targets the only relevant segments, processing them in parallel.
+Each shard is divided into **segments**. They are a smaller storage unit within a shard, storing a subset of vectors and their associated payloads (metadata). When a query is executed, it targets only the relevant segments, processing them in parallel.
-
+
### Replication: High Availability and Data Integrity
@@ -396,7 +433,7 @@ You don’t want a single failure to take down your system, right? Replication k
In Qdrant, **Replica Sets** manage these copies of shards across different nodes. If one replica becomes unavailable, others are there to take over and keep the system running. Whether the data is local or remote is mainly influenced by how you've configured the cluster.
-
+
When a query is made, if the relevant data is stored locally, the local shard handles the operation. If the data is on a remote shard, it’s retrieved via gRPC.
@@ -413,17 +450,15 @@ client.create_collection(
We recommend using sharding and replication together so that your data is both split across nodes and replicated for availability.
-For more details on features like **user-defined sharding, node failure recovery**, and **consistency guarantees**, see our guide on [Distributed Deployment.](https://qdrant.tech/documentation/distributed_deployment/)
+For more details on features like **user-defined sharding, node failure recovery**, and **consistency guarantees**, see our guide on [Distributed Deployment.](https://qdrant.tech/documentation/scaling/distributed_deployment/)
## Multitenancy: Data Isolation for Multi-Tenant Architectures
-
-
Sharding efficiently distributes data across nodes, while replication guarantees redundancy and fault tolerance. But what happens when you’ve got multiple clients or user groups, and you need to keep their data isolated within the same infrastructure?
**Multitenancy** allows you to keep data for different tenants (users, clients, or organizations) isolated within a single cluster. Instead of creating separate collections for `Tenant 1` and `Tenant 2`, you store their data in the same collection but tag each vector with a `group_id` to identify which tenant it belongs to.
-
+
In the backend, Qdrant can store `Tenant 1`’s data in Shard 1 located in Canada (perhaps for compliance reasons like GDPR), while `Tenant 2`’s data is stored in Shard 2 located in Germany. The data will be physically separated but still within the same infrastructure.
@@ -449,7 +484,9 @@ If you want to learn more about working with a multitenant setup in Qdrant, you
A common security risk in vector databases is the possibility of **embedding inversion attacks**, where attackers could reconstruct the original data from embeddings. There are many layers of protection you can use to secure your instance that are very important before getting your vector database into production.
-For quick security in simpler use cases, you can use the **API key authentication**. To enable it, set up the API key in the configuration or environment variable.
+> Self-hosted open source deployments are not secure by default and are not production-ready. Qdrant Cloud deployments are always secure and production-ready.
+
+For quick security in simpler use cases of your self-hosted instances, you can use the **API key authentication**. To enable it, set up the API key in the configuration or environment variable.
```yaml
service:
@@ -472,11 +509,11 @@ In more advanced setups, Qdrant uses **JWT (JSON Web Tokens)** to enforce **Role
RBAC defines roles and assigns permissions, while JWT securely encodes these roles into tokens. Each request is validated against the user's JWT, ensuring they can only access or modify data based on their assigned permissions.
-You can easily setup your access tokens and secure access to sensitive data through the **Qdrant Web UI:**
+You can easily set up your access tokens and secure access to sensitive data through the **Qdrant Web UI:**
-
+
-By default, Qdrant instances are **unsecured**, so it's important to configure security measures before moving to production. To learn more about how to configure security for your Qdrant instance and other advanced options, please check out the [official Qdrant documentation on security.](https://qdrant.tech/documentation/security/)
+By default, self-hosted Qdrant instances are **unsecure**, so it's important to configure security measures before moving to production. To learn more about how to configure security for your Qdrant instance and other advanced options, please check out the [official Qdrant documentation on security.](https://qdrant.tech/documentation/security/)
## Time to Experiment
@@ -484,21 +521,6 @@ As we've seen in this article, a vector database is definitely not **just** a da
But there’s no better way to learn than by doing. Try building a [semantic search engine](https://qdrant.tech/documentation/tutorials/search-beginners/) or experiment deploying a [hybrid search service](https://qdrant.tech/documentation/tutorials/hybrid-search-fastembed/) from zero. You'll realize there are endless ways you can take advantage of vectors.
-| **Use Case** | **How It Works** | **Examples** |
-|-----------------------------------|------------------------------------------------------------------------------------------------------|-----------------------------------------------------------|
-| **Similarity Search** | Finds similar data points using vector distances | Find similar product images, retrieve documents based on themes, discover related topics |
-| **Anomaly Detection** | Identifies outliers based on deviations in vector space | Detect unusual user behavior in banking, spot irregular patterns |
-| **Recommendation Systems** | Uses vector embeddings to learn and model user preferences | Personalized movie or music recommendations, e-commerce product suggestions |
-| **RAG (Retrieval-Augmented Generation)** | Combines vector search with large language models (LLMs) for contextually relevant answers | Customer support, auto-generate summaries of documents, research reports |
-| **Multimodal Search** | Search across different types of data like text, images, and audio in a single query. | Search for products with a description and image, retrieve images based on audio or text |
-| **Voice & Audio Recognition** | Uses vector representations to recognize and retrieve audio content | Speech-to-text transcription, voice-controlled smart devices, identify and categorize sounds |
-| **Knowledge Graph Augmentation** | Links unstructured data to concepts in knowledge graphs using vectors | Link research papers to related studies, connect customer reviews to product features, organize patents by innovation trends|
-
-
-You can also watch our video tutorial and get started with Qdrant to generate semantic search results and recommendations from a sample dataset.
-
-
-
Phew! I hope you found some of the concepts here useful. If you have any questions feel free to send them in our [Discord Community](https://discord.com/invite/qdrant) where our team will be more than happy to help you out!
> Remember, don't get lost in vector space! 🚀
diff --git a/qdrant-landing/content/articles/what-is-quantization.md b/qdrant-landing/content/articles/what-is-quantization.md
index b5df834f1..7a4fd7193 100644
--- a/qdrant-landing/content/articles/what-is-quantization.md
+++ b/qdrant-landing/content/articles/what-is-quantization.md
@@ -255,7 +255,7 @@ POST /collections/{collection_name}/points/search
```python
client.query_points(
collection_name="my_collection",
- query_vector=[0.22, -0.01, -0.98, 0.37], # Your query vector
+ query=[0.22, -0.01, -0.98, 0.37], # Your query vector
search_params=models.SearchParams(
quantization=models.QuantizationSearchParams(
rescore=True # Enables rescoring with original vectors
@@ -347,7 +347,7 @@ POST /collections/{collection_name}/points/search
```python
client.query_points(
collection_name="my_collection",
- query_vector=[0.22, -0.01, -0.98, 0.37],
+ query=[0.22, -0.01, -0.98, 0.37],
search_params=models.SearchParams(
quantization=models.QuantizationSearchParams(
rescore=True, # Enables rescoring with original vectors
diff --git a/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md b/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md
new file mode 100644
index 000000000..4d24658fc
--- /dev/null
+++ b/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md
@@ -0,0 +1,190 @@
+---
+title: "When Is a Reranker Worth It?"
+short_description: "Rerank 10 candidates, compare with your tuned first stage on held-out queries, and raise the count only after the win holds."
+description: "Test whether a cross-encoder reranker beats your tuned first stage in Qdrant, then choose the model and candidate count from measured results."
+preview_dir: /articles_data/when-a-reranker-is-worth-it/preview
+social_preview_image: /articles_data/when-a-reranker-is-worth-it/preview/social_preview.jpg
+weight: -210
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-23T00:00:00+03:00
+draft: false
+keywords:
+ - cross-encoder reranker
+ - reranking
+ - MMR
+ - search relevance
+ - FastEmbed
+category: search-quality
+---
+
+Before you tune a reranker, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline.
+
+Your candidate list can already contain documents your ranking never shows. Score those candidates as if they were perfectly ordered, then compare that with the score your pipeline returns today. The gap between the two is everything a better ranking stage could recover, so measure it before you reach for a model. Use `nDCG@10`, which grades the top 10 results and gives more credit to relevant documents near the top.
+
+A wide gap means a better order is worth chasing. The deepest count measured here was 200 candidates. At that count, the `nDCG@10` gap ran from 0.247 to 0.487 across the five datasets, and [candidate depth](/articles/candidate-depth/) shows how to measure it on your own collection.
+
+Reranking covers several model families. This article measures cross-encoders, which read the query and candidate together as a single sequence. A classification head returns one relevance score for the pair. Joint reading lets the model capture token interactions that separately encoded query and document vectors miss.
+
+Every candidate takes a forward pass at query time, which rules a cross-encoder out as a first stage and keeps it in the reranking slot. Late interaction models rerank from stored vectors instead, and the last section covers where they fit.
+
+
+
+## Test a Reranker in Three Steps
+
+1. Establish the baseline the reranker has to beat: [tuned fusion](/articles/how-to-tune-hybrid-search/) if you run hybrid search, your current ranking if you run dense-only or sparse-only. Confirm the documents your labels mark relevant reach the candidate list. A reranker only reorders what it receives; missing documents are a [candidate depth](/articles/candidate-depth/) or retrieval problem.
+2. Rerank 10 candidates with the model you would actually serve, since model choice moved our results more than any other setting. Read its model card first for languages, domains, and context window. [Reranking with FastEmbed](/documentation/fastembed/fastembed-rerankers/) shows the cross-encoder workflow and the available models. Compare the result with the first-stage baseline on held-out labeled queries.
+3. Raise the candidate count only if the reranker wins. Measure throughput on your document lengths before making it part of the serving path.
+
+Step 2 starts from the request your service already sends. Request the payload fields the reranker reads and your labels key on, and keep the fusion settings you serve today.
+
+```python
+from qdrant_client import QdrantClient, models
+
+# Both prefetches use the models the collection was indexed with.
+from your_embedding_setup import dense_query, sparse_query
+
+client = QdrantClient(
+ url="https://YOUR-CLUSTER.cloud.qdrant.io",
+ api_key="",
+)
+query_text = "the query text"
+
+fused = client.query_points(
+ collection_name="products",
+ prefetch=[
+ models.Prefetch(query=dense_query, using="dense", limit=200),
+ models.Prefetch(query=sparse_query, using="bm25", limit=200),
+ ],
+ # Your tuned fusion settings; k=2 with equal weights is the default.
+ # RrfQuery needs Qdrant v1.17 or later and a client release that exposes it.
+ query=models.RrfQuery(rrf=models.Rrf(k=2, weights=[1.0, 1.0])),
+ limit=10,
+ with_payload=["text", "doc_id"],
+).points
+```
+
+Then score the same 10 candidates with a cross-encoder and sort them by that score. Any [FastEmbed cross-encoder](/documentation/fastembed/fastembed-rerankers/) works here, so name the model you plan to serve. `Xenova/ms-marco-MiniLM-L-6-v2` appears in the examples because it is the quickest to download.
+
+```python
+from fastembed.rerank.cross_encoder import TextCrossEncoder
+
+encoder = TextCrossEncoder(model_name="Xenova/ms-marco-MiniLM-L-6-v2")
+scores = list(encoder.rerank(query_text, [point.payload["text"] for point in fused]))
+
+# Sort by position so tied scores never compare the points themselves.
+order = sorted(range(len(fused)), key=lambda i: scores[i], reverse=True)
+reranked = [fused[i] for i in order]
+```
+
+You now have two orderings of the same 10 candidates. Score both with the `nDCG@10` function from the [pre-tuning article](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain).
+
+```python
+# relevance holds this query's labels, keyed by doc_id.
+before = ndcg_at_k([point.payload["doc_id"] for point in fused], relevance)
+after = ndcg_at_k([point.payload["doc_id"] for point in reranked], relevance)
+```
+
+Run that over your labeled queries, average the per-query difference, then check the interval around it before you trust the direction.
+
+## Compare with the Best First Stage
+
+Compare the reranker against the strongest first stage you can build. Qdrant's default reciprocal rank fusion (RRF) is already a solid baseline, and fusion tuned on your own labels is stronger. A reranker measured against the default can look like a win that tuning would have delivered for far less work at query time.
+
+So [tune fusion](/articles/how-to-tune-hybrid-search/) first, then make that tuned ranking the number the reranker has to beat on held-out labeled queries.
+
+Each row in the following table reports the best of four cross-encoders on that dataset. The deltas show the `nDCG@10` change over default RRF and over fusion tuned on the same candidates. `MiniLM-L-6`, `MiniLM-L-12`, and `bge-reranker-base` truncate each pair at 512 tokens. `jina-reranker-v2` reads up to 1024 and was trained on a broader mix, including code.
+
+The held-out column is the one that decides. It holds the share of 200 split-half draws where the gain survived on queries it was not selected on.
+
+| Dataset | Best Model | vs. Default RRF | vs. Tuned Fusion | Held Out |
+|---|---|---|---|---|
+| SciFact | `jina-reranker-v2` @ 200 | +0.057 | +0.033 | no, 37% |
+| ArguAna | `jina-reranker-v2` @ 25 | +0.031 | +0.017 | no, 2.5% |
+| WANDS | `MiniLM-L-6` @ 200 | +0.039 | -0.008 | no, 0% |
+| CodeSearchNet | `jina-reranker-v2` @ 200 | +0.169 | +0.135 | yes, 100% |
+| DBPedia-entity | `jina-reranker-v2` @ 200 | +0.137 | +0.115 | yes, 100% |
+
+Ship a reranker gain only when it survives held-out validation. The two confirmed wins here held in 100% of the split-half draws, while the three unconfirmed results held in under half of them. When a positive result fails the split, add queries or keep fusion, and use [the label-count table in the pre-tuning article](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to size that confirmation.
+
+Keep fusion when the reranker loses to the tuned first stage. WANDS gained +0.039 over default RRF and still lost to fusion tuned on the same candidates.
+
+Both confirmed wins came from the model whose window and training data fit the corpus, scoring the same candidates the other three saw. Even those wins closed part of the gap, recovering 46% of it on CodeSearchNet and 24% on DBPedia-entity.
+
+On DBPedia-entity, fusion left the page for "(Just Like) Starting Over" at rank 49 of 200, even though every term of the query "John Lennon Yoko Ono album Starting Over" sits in the page's first two sentences. Fusion reads ranks, and neither prefetch ranked the page high: 35th dense, 45th sparse, behind pages whose titles name both Lennon and Ono. The cross-encoder read the query and page as one sequence and put it first.
+
+## Diagnose a Loss Before You Stop
+
+If the reranker loses at 10 candidates, go back to fit and measure it. Tokenize a sample of your query-document pairs with the model's tokenizer. Compare the 95th percentile length with the model's window: a longer pair gets truncated, and the model scores a document it only partly read. Then reread the card's languages and domains against your corpus.
+
+Two mismatches explain every loss in the table. Long queries are the first. ArguAna queries average 168 words, which leaves little room for the document inside a 512-token truncation. The 1024-token model turned that loss into a win, though it held in only 2.5% of the held-out splits, so the label set cannot confirm it.
+
+A training-domain gap is the second. None of the three older models was trained on code, and `jina-reranker-v2` flipped CodeSearchNet from a loss into the largest confirmed win in the table, scoring the same candidates the others saw.
+
+If you find a mismatch, swap in a model whose window and training data fit your documents and rerun the 10-candidate test. If the model fits and still loses, keep the tuned first stage and spend the tuning effort elsewhere: on WANDS, tuned fusion beat all four models at every candidate count.
+
+## Set Candidate Count After a Win
+
+Start with 10 candidates, and confirm on your labeled queries that the reranker beats tuned fusion before you change the count. Every configuration that trailed tuned fusion at 10 candidates still trailed it at 200, so a deeper list does not rescue a reranker that loses at 10. `nDCG@10` grades the same top 10 results at every count, so the count changes only what the reranker gets to choose from.
+
+
+
+_The best nDCG@10 change over tuned fusion among the four models, by candidate count. A line above zero is a reranker win; WANDS never crosses it._
+
+Step 1 confirmed that your relevant documents reach the candidate list. Run that same check at each count you are considering, before you run the reranker at any of them. The share of queries whose relevant documents are already in the candidate list limits how much increasing the count can help. Beyond that point, extra candidates only add documents that can push the relevant ones out of the top 10.
+
+Rerank only at candidate counts where that share is still climbing. When the first stage finds relevant documents early, the share flattens quickly. In ArguAna, each query has one relevant document, and 90% of queries already included it within the first 25 candidates. Increasing the count to 200 raised that share to 98%, but turned the gain into a loss.
+
+Where the share keeps rising, deeper reranking keeps finding documents the first stage buried. CodeSearchNet held the relevant document for 80.5% of queries at 25 candidates and for 90.5% at 200, and its gain kept climbing through 200.
+
+The relevance gain can flatten before that share does, so take the smallest count that captures most of it. DBPedia-entity reached 96% of its eventual gain by 50 candidates, so going to 200 quadrupled the reranking work for the last 4%.
+
+The shape you get depends on how many relevant documents your queries have and how well your first stage already ranks them. Measure it on your own labels rather than borrowing a count from these datasets.
+
+## Size Reranking for Production
+
+Once relevance has settled the candidate count and the model, measure query-candidate pairs per second and tail latency on the hardware you plan to deploy, using representative document lengths and concurrency.
+
+The table shows CPU throughput for the four [FastEmbed cross-encoders](/documentation/fastembed/fastembed-rerankers/), listed by their full model IDs and measured in one process on an Apple M5 Pro with 15 threads. The last column converts that rate to whole queries at 100 candidates each.
+
+| Model | Size | Docs per Second | Queries per Second |
+|---|---|---|---|
+| `Xenova/ms-marco-MiniLM-L-6-v2` | 0.08 GB | 64 to 212 | 0.6 to 2.1 |
+| `Xenova/ms-marco-MiniLM-L-12-v2` | 0.12 GB | 34 to 117 | 0.3 to 1.2 |
+| `BAAI/bge-reranker-base` | 1.04 GB | 16 to 45 | 0.2 to 0.5 |
+| `jinaai/jina-reranker-v2-base-multilingual` | 1.11 GB | under 2 | under 0.02 |
+
+Document length explains each range. DBPedia-entity has short entity abstracts, while SciFact has full paper abstracts.
+
+Weigh those rates against the held-out gain. At 100 candidates, one CPU process spends between half a second and five seconds per query with the three smaller models, where the second prefetch behind [tuned fusion](/articles/how-to-tune-hybrid-search/) added 0.6 to 1.5 ms in the same setup. The 10-candidate test itself stays fast even on CPU, at 47 to 156 ms per query with the smallest model, so run it before you plan any serving work.
+
+Pick the model on fit rather than size. `bge-reranker-base` and `jina-reranker-v2` are nearly the same size, and only the second ever beat tuned fusion. Training data and context window separated them.
+
+Some models only reach a usable rate on a GPU. The `jina-reranker-v2` ONNX export runs one CPU thread at a time through an attention kernel, which is why its row reads under 2. On a GPU through PyTorch it ran at 32 to 310 documents per second, and the quality numbers here come from that run. It ships under a CC-BY-NC-4.0 license, so check the terms first.
+
+## Use Other Stages for Different Problems
+
+Match the stage to the symptom you see in your results.
+
+| Symptom | Stage |
+|---|---|
+| Relevant candidates ranked below weaker ones | A cross-encoder or a late interaction model as a reranker |
+| Results are repetitive or near-duplicates | [Maximal marginal relevance](/documentation/search/search-relevance/#maximal-marginal-relevance-mmr) |
+| One document's chunks fill the first page | [Grouping](/documentation/search/search/#grouping-api) |
+| Recency, popularity, or other payload signals should shape the order | [Formula Query](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) |
+
+Maximal marginal relevance trades relevance for diversity, and `nDCG` does not reward the diversity it adds, so measure the direction on your own labels before shipping it.
+
+Grouping fits collections that store each chunk of a document as its own point. `query_points_groups` with `group_by` on the document ID field returns the best chunk per document, so one long document cannot fill the first page. The grouped field needs a [payload index](/documentation/manage-data/indexing/#payload-index); without one on `document_id`, Qdrant Cloud returns a 400.
+
+Formula Query rescores the same candidates with an expression over payload fields, such as recency or popularity, and needs a payload index on each field the formula references.
+
+Test a [late interaction model](/documentation/fastembed/fastembed-colbert/) when a cross-encoder is too slow. Document vectors are built at ingest, so only the query goes through the model per request, and the rescoring runs inside Qdrant in the same multi-stage query. Storage grows to a vector per token. [Multivectors and Late Interaction](/documentation/tutorials-search-engineering/using-multivector-representations/) walks the full setup.
+
+## What to Tune Next
+
+After a win, the work moves to throughput, where the candidate count you can serve decides how much of the gain survives. After a loss, the gap is still there and the candidates are what to change, so revisit retrieval and [candidate depth](/articles/candidate-depth/) before adding another ranking stage.
+
+Next, if memory is the constraint, [measure what memory placement and rescoring add to query latency](/articles/when-your-collection-outgrows-ram/).
diff --git a/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md b/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md
new file mode 100644
index 000000000..6f8732b47
--- /dev/null
+++ b/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md
@@ -0,0 +1,203 @@
+---
+title: "When Your Collection Outgrows RAM"
+short_description: "Keep the quantized copy in RAM and the original vectors on disk, then measure what rescoring reads back on your own deployment."
+description: "Set quantization and memory placement in Qdrant once a collection outgrows RAM: what the rescoring disk read costs and what quality it recovers."
+preview_dir: /articles_data/when-your-collection-outgrows-ram/preview
+social_preview_image: /articles_data/when-your-collection-outgrows-ram/preview/social_preview.jpg
+weight: -209
+author: Dylan Couzon
+author_link: https://www.linkedin.com/in/dcouzon/
+date: 2026-08-24T00:00:00+03:00
+draft: false
+keywords:
+ - memory tiers
+ - quantization
+ - rescoring
+ - oversampling
+ - TurboQuant
+category: search-quality
+---
+
+Once a collection no longer fits in RAM, the kernel evicts vector pages, and the next query waits on a disk read to get them back. Quantization buys that memory back. Qdrant keeps a compressed copy of each dense vector in RAM and moves the full-precision originals to disk.
+
+[TurboQuant](/documentation/manage-data/quantization/#turboquant-quantization) is the method measured here. It rotates each vector before compressing it, which spreads the error evenly across coordinates, and its `bits` parameter sets the depth from `bits4` down to `bits1`. Start at `bits4`, a good default for many workloads at eight times compression.
+
+A lower-precision [datatype](/documentation/manage-data/vectors/#datatypes) such as `float16` shrinks the same vectors a different way. Quantization adds a compressed copy beside the originals, while a datatype changes the originals themselves, and that difference decides whether anything full-precision survives to rescore against.
+
+Dense vectors take most of that memory in a single-vector collection. If you use a late interaction model, its multivectors dominate instead, at one vector per token.
+
+Every measurement below comes from a dense-only request. In hybrid search the dense and sparse reads share one page cache, so rerun your full query before you size a deployment or set a latency budget.
+
+
+
+## Set Quantization in Four Steps
+
+1. Estimate the resident footprint of your dense vectors, and how far it overshoots the memory you have.
+2. Choose the quantization method and bit depth, starting from `bits4`.
+3. Pin the quantized copy and leave the original vectors `cold`.
+4. Set `rescore` and `oversampling` from measurements on the deployment you will serve.
+
+For step 1, this formula estimates the RAM needed to keep all `float32` dense vectors resident:
+
+```text
+RAM = number of vectors × vector dimensions × 4 bytes × 1.5
+```
+
+The extra 50% covers metadata, indexes, point versions, and temporary segments created during optimization. Treat the result as a starting estimate, not a container limit. For a full estimate with payloads, indexes, and replication, use the [Qdrant Sizing Calculator](https://sizing.qdrant.tech/).
+
+Qdrant [recommends pinning the quantized copy with `cold` originals](/documentation/manage-data/quantization/#memory-and-speed-tuning) to shrink the footprint while keeping search fast. The following two sections measure what that pairing costs in disk reads and what rescoring recovers.
+
+Step 4 needs a [labeled set](/articles/before-tuning-a-qdrant-collection/). Compare `nDCG@k` with `k` set to the number of results you return, pick the configuration on one part of the set, then confirm it on queries that took no part in the selection. Use `Recall@k` against exact search to explain a loss.
+
+## Rescoring Adds the Disk Read
+
+`rescore` repairs part of the error that compression introduces. Qdrant reads the original vectors back after the dense prefetch and reorders the top candidates by their full-precision scores.
+
+`oversampling` sets how many candidates Qdrant pre-selects for that pass. At `oversampling` 2 with a limit of 10, the prefetch collects 20 candidates from the quantized copy, scores them against the originals, and returns the best 10.
+
+Both are query parameters, so a request can change them without touching the collection. Once the originals live on disk, each of those rereads is a disk read.
+
+Since v1.19, Qdrant sets [memory placement](/documentation/ops-configuration/memory-tiers/) per structure with `memory`, replacing the deprecated `on_disk` and `always_ram` flags. Data moves between disk and RAM in fixed-size pages, typically 4 KiB on Linux, and the placement decides where a structure's pages sit.
+
+- `cold` loads lazily from disk, so the first request that needs a page waits for it.
+- `cached` enters the page cache when the collection loads, and the kernel may evict it later.
+- `pinned` stays in RAM, so the structure has to fit.
+
+Only the quantized copy can be pinned. Qdrant reads the [original vectors through a memory map](/documentation/ops-configuration/memory-tiers/#limitations), so they take `cold` or `cached`. Set both placements explicitly, because the quantized copy defaults to following the originals.
+
+In hybrid search, budget for the [sparse vector index](/documentation/ops-configuration/memory-tiers/#sparse-vector-index) too: it takes the same placements and defaults to `pinned`, holding RAM the quantized copy needs.
+
+The memory cap decides what those placements deliver. The same query took about 4 ms with the originals resident, and 43 ms when rescoring reread them under a 4 GiB limit.
+
+| Limit | Original Vectors | Quantized Vectors | `rescore` | p50 ms | GB Read, Both Passes |
+|---|---|---|---|---|---|
+| 12 GiB | `cached` | `pinned` | off | 3.8 | 0.30 |
+| 12 GiB | `cached` | `pinned` | on | 4.1 | 0.52 |
+| 4 GiB | `cached` | `pinned` | off | 4.3 | 0.30 |
+| 4 GiB | `cached` | `pinned` | on | 43.4 | 2.98 |
+| 4 GiB | `cold` | `cached` | on | 45.7 | 3.02 |
+| 4 GiB | `cold` | `pinned` | on | 52.0 | 3.50 |
+
+Rescoring is nearly free while the originals stay in cache, and it becomes the slowest part of the query once they do not. At 12 GiB it added 0.3 ms. At 4 GiB it added 39.1 ms.
+
+Neither the cap nor the rescoring pass causes it alone: with rescoring off, the query ran within half a millisecond of itself at both limits. The penalty is rereading original-vector pages rather than scoring candidates, and at 4 GiB rescoring read 2.98 GB instead of 0.30 GB. Moving the originals from `cached` to `cold` left the median inside its own run-to-run spread, so placement does not remove those reads. Test both settings under your own container limit to see what rescoring costs you.
+
+Two changes reduce that read. More memory keeps the originals resident, and a lower `oversampling` rereads fewer candidates without needing any. At `bits1`, the candidates past the first rescoring pass buy little quality, which the next section measures.
+
+Set placement for the footprint instead. Pin the quantized copy, which is a fraction of the originals' size and fits where they cannot, and leave the originals `cold`.
+
+Async I/O then makes those `cold` reads cheaper without more memory. Set [`storage.performance.io_uring` to `auto`](/documentation/ops-configuration/memory-tiers/#async-io) in the configuration file, and Qdrant issues a query's rereads together and waits for them in parallel rather than one after another. It is disabled by default, covers `cold` structures only, and needs a Linux kernel that supports io_uring.
+
+
+
+## What Rescoring Recovers
+
+Two measurements answer different questions here. An exact search scans every vector and gives the reference result, while graph search trades some of those neighbors for lower latency.
+
+`Recall@10` against exact search is the share of the exact top 10 that a configuration returned, so it reports what happened inside the dense prefetch. `nDCG@10` grades the returned top 10 against DBPedia-entity's labels, giving more credit to relevant documents near the top, so it reports what the user sees.
+
+We picked a candidate on a separate labeled set. The rule was the lowest `oversampling` and lowest `bits` value staying within 0.01 `nDCG@10` of float32, and within 0.02 `Recall@10`.
+
+The table reports how each configuration then scored on 200 held-out queries.
+
+
+
+| Quantization | `rescore` | `nDCG@10` | `Recall@10` Against Exact |
+|---|---|---|---|
+| float32 | not applicable | 0.3103 | 0.957 |
+| TurboQuant `bits4` | off | 0.3218 | 0.918 |
+| TurboQuant `bits4` | on, `oversampling` 4 | 0.3238 | 0.993 |
+| TurboQuant `bits1` | off | 0.2786 | 0.605 |
+| TurboQuant `bits1` | on, `oversampling` 1 | 0.3114 | 0.951 |
+| TurboQuant `bits1` | on, `oversampling` 2 | 0.3128 | 0.977 |
+| TurboQuant `bits1` | on, `oversampling` 4 | 0.3178 | 0.988 |
+
+Measure float32 at your own graph-search settings first, so you can separate what the graph misses from what quantization costs. Here it returned 0.957 `Recall@10`, so approximate traversal missed roughly 4% of the exact top 10 before quantization entered the comparison.
+
+What rescoring recovers depends on how much precision the bit depth discarded. At `bits4` it lifted dense-prefetch `Recall@10` from 0.918 to 0.993, while 200 held-out queries did not establish an `nDCG@10` difference. Recovered neighbors can improve dense-prefetch recall without improving the labeled top 10.
+
+At a deep bit depth, rescoring is what makes the quantization usable. One pass raised `bits1` from 0.605 to 0.951 `Recall@10`. Qdrant [enables `rescore` by default](/documentation/manage-data/quantization/#searching-with-quantization) for `bits1`, `bits1_5`, `bits2`, and binary quantization for this reason.
+
+
+
+_One rescoring pass does most of the recovery at bits1. Raising oversampling past 1 buys little, which is why the disk reads it adds are the cost to watch._
+
+After `oversampling` 1, extra candidates add disk reads for little recall. `bits1` reached 0.977 `Recall@10` at `oversampling` 2 and 0.988 at `oversampling` 4.
+
+The selection rule picked `bits1` with `rescore` and `oversampling` 1. Against float32 on the held-out queries, its `nDCG@10` came in 0.0011 higher, with a paired 95% interval from -0.003 to +0.005.
+
+Read that interval as a bound rather than proof of an identical ranking. On this dataset it holds the `nDCG@10` difference within 0.005 either way, and `Recall@10` came in 0.006 lower.
+
+## If TurboQuant Is Not the Right Fit
+
+For a comparison of TurboQuant bit depths across ten datasets, see [TurboQuant in Qdrant](/articles/turboquant-quantization/). Qdrant also supports [Scalar, Binary, and Product Quantization](/documentation/manage-data/quantization/), and each one validates with the same dense-prefetch and held-out checks.
+
+- [Scalar Quantization](/documentation/manage-data/quantization/#scalar-quantization): converts vector components to `int8`. Start here when moderate compression is enough.
+
+- [Binary Quantization](/documentation/manage-data/quantization/#binary-quantization): a compact, fast option that works best with high-dimensional embeddings whose components have a centered distribution. Measure whether rescoring recovers enough quality for your workload.
+
+- [Product Quantization](/documentation/manage-data/quantization/#product-quantization): prioritizes a smaller memory footprint, with a larger accuracy and search-speed trade-off to validate.
+
+### `turbo4` Changes What Rescoring Reads
+
+The [`turbo4` datatype](/documentation/manage-data/vectors/#turbo4) stores each dense vector as 4 bits per dimension, about one-eighth of its original size. It is built on TurboQuant, and it replaces the full-precision vector rather than sitting beside it.
+
+Rescoring still works on top of it. Pairing `turbo4` with 1-bit TurboQuant searches the compact index and rescores against the 4-bit vectors, which costs less storage than 1-bit over full precision and gives up some rescoring precision.
+
+Keep full-precision vectors when you want rescoring at the accuracy this article measures. This article does not measure `turbo4`, so validate it on your own queries and labels.
+
+## Verify It on Your Own Collection
+
+Configure the bit depth and the two placements on the collection you already have, using the name of your dense vector. `rescore` and `oversampling` belong on the query, which is what makes them cheap to compare.
+
+```python
+from qdrant_client import QdrantClient, models
+
+client = QdrantClient(
+ url="https://YOUR-CLUSTER.cloud.qdrant.io",
+ api_key="",
+)
+
+client.update_collection(
+ # Replace with the collection that contains your dense vector.
+ collection_name="products",
+ quantization_config=models.TurboQuantization(
+ turbo=models.TurboQuantQuantizationConfig(
+ # Replace with the bit depth selected by your evaluation.
+ bits=models.TurboQuantBitSize.BITS1,
+ memory=models.Memory.PINNED,
+ )
+ ),
+ vectors_config={"dense": models.VectorParamsDiff(memory=models.Memory.COLD)},
+)
+```
+
+First, compute the exact dense top `k` once for a representative sample of your queries. This article reports `k=10`. An exact search reads every original vector, so keep the sample small enough for the cost you can accept.
+
+Then run your existing dense prefetch with each `rescore` and `oversampling` variant, changing nothing else. `Recall@k` against the exact result shows what quantization changed in the dense prefetch. Without labels, that check and the latency numbers still stand on their own.
+
+For hybrid search, keep the prefetches, [fusion settings](/articles/how-to-tune-hybrid-search/), and filters your service already uses, then compare the final `nDCG@k`.
+
+### Self-Hosted
+
+Run the dense-prefetch check under the memory cap you deploy with, from a cold page cache followed by a measured pass. Run `rescore=False` even if you would never ship it, because it shows the cost of the rest of the dense prefetch.
+
+### Qdrant Cloud
+
+Measure the full request under its normal operating conditions. The cluster sets the container limit and the page-cache state, which leaves the placements and the query parameters as what you compare.
+
+## What to Tune Next
+
+Keep the first configuration that meets your held-out `nDCG@k` and latency targets. If none qualifies, test another quantization method or a higher `oversampling` value.
+
+On a multi-shard hybrid collection, rerun the full request on your deployed shard layout once the dense-vector placements are set. Each shard runs the prefetch and rescoring against its own data.
+
+With a `limit` of 200 and `oversampling` 1, rescoring can read up to 200 original vectors per shard, or up to 2,400 across 12 shards. [Candidate depth](/articles/candidate-depth/) covers how to set the limit that total scales with.
+
+If you do not have a labeled query set yet, [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/) covers how to build one.
diff --git a/qdrant-landing/content/blog/case-study-and-ai.md b/qdrant-landing/content/blog/case-study-and-ai.md
index ea735079a..eda840ebb 100644
--- a/qdrant-landing/content/blog/case-study-and-ai.md
+++ b/qdrant-landing/content/blog/case-study-and-ai.md
@@ -24,13 +24,25 @@ partition: case-studies
[&AI](https://tryandai.com/) is on a mission to redefine patent litigation. Their platform helps legal professionals invalidate patents through intelligent prior art search, claim charting, and automated litigation support. To make this work at scale, CTO and co-founder Herbie Turner needed a vector database that could power fast, accurate retrieval across billions of documents without ballooning DevOps complexity. That’s where Qdrant came in.
+{{< quote
+ text="With Qdrant, we scaled to a billion vectors and still respond in sub-second latency. That lets us power workflows that used to take hours in just a few minutes."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI"
+ logo="/img/customers-case-studies-logo/and-ai.svg"
+ featured="true" >}}
+
## Legal tech’s toughest retrieval challenge
Patent litigation is a high-stakes game. When a company is sued for patent infringement, the best defense is often to invalidate the patent altogether. That means proving the idea was disclosed publicly before the patent was granted. Finding that “prior art” requires sifting through vast, multilingual document corpora with domain-specific technical language.
Traditionally, this is done through outsourced search firms or attorneys running boolean queries across multiple databases. It’s time-consuming, expensive, and heavily reliant on human intuition. Turner and co-founder Caleb Harris saw an opportunity to use modern AI tooling and large language models (LLMs) to reframe the problem.
-"Instead of generating legal text, which attorneys rightly distrust, we focused everything around retrieval," said Turner. "If we can ground our results in real documents, hallucination risk is minimized."
+{{< quote
+ text="Instead of generating legal text, which attorneys rightly distrust, we focused everything around retrieval. If we can ground our results in real documents, hallucination risk is minimized."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
## A retrieval-first legal AI stack
@@ -41,12 +53,20 @@ From the start, \&AI framed patent invalidation and charting as semantic retriev
But the scale was immense. Their full corpus includes hundreds of millions of documents from international patent offices and other sources, resulting in more than 250 billion tokens. Ingesting, embedding, and searching this volume of data demanded a robust, cloud-native vector search solution.
-"We needed to scale to a number of vectors that just hadn’t been benchmarked publicly," said Turner. "Qdrant was the only one that handled that load out of the box — and without needing dedicated DevOps engineers."
+{{< quote
+ text="We needed to scale to a number of vectors that just hadn’t been benchmarked publicly. Qdrant was the only one that handled that load out of the box — and without needing dedicated DevOps engineers."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI"
+ logo="/img/customers-case-studies-logo/and-ai.svg" >}}
Turner had used Qdrant in a prior startup, where he appreciated the high performance and strong Rust-based architecture. But it was Qdrant’s [opinionated documentation](https://qdrant.tech/documentation/) and built-in developer tools that sealed the deal.
-*“I’m all for opinionated docs,” said Turner. “Don’t make me figure out how to optimize everything myself. Qdrant tells you the right way to do things; it just works.”*
-— Herbie Turner, CTO & Co-Founder, \&AI
+{{< quote
+ text="I’m all for opinionated docs. Don’t make me figure out how to optimize everything myself. Qdrant tells you the right way to do things; it just works."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
## From noisy PDFs to structured vectors
@@ -60,22 +80,32 @@ They chose [scalar quantization](https://qdrant.tech/articles/scalar-quantizatio
Rather than rely on LLMs to generate legal output, \&AI framed its tasks as retrieval problems. Everything, prior art search, invalidity charts, claim comparisons, was treated as a ranking and grounding problem.
-"We do an initial broad search to get candidates, then use metadata filtering, claim construction analysis, and context-specific re-ranking to refine results," said Turner.
+{{< quote
+ text="We do an initial broad search to get candidates, then use metadata filtering, claim construction analysis, and context-specific re-ranking to refine results."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
Qdrant’s filterable HNSW, payload field indexing, and support for multi-tenancy made this possible. Public patent search operates globally, while firm-specific legal data is stored in isolated tenant spaces.
-"Having multi-tenancy built-in was huge," Turner said. "It let us give firms strong guarantees around data privacy without spinning up separate infrastructure."
+{{< quote
+ text="Having multi-tenancy built-in was huge. It let us give firms strong guarantees around data privacy without spinning up separate infrastructure."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
## Scaling infrastructure, not headcount
By using [Qdrant Cloud](https://qdrant.tech/cloud/), \&AI avoided the need to manage DevOps or self-host massive vector clusters. Even after scaling to over 1 billion vectors, Qdrant’s managed infrastructure delivered fast search and low memory usage.
-"Patent litigation has huge stakes, one result could influence a billion-dollar case," said Turner. "Accuracy is the top priority, and Qdrant let us optimize for that without compromising on cost or performance."
+{{< quote
+ text="Patent litigation has huge stakes, one result could influence a billion-dollar case. Accuracy is the top priority, and Qdrant let us optimize for that without compromising on cost or performance."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
Qdrant’s support for [payload filters](https://qdrant.tech/documentation/search/filtering/), [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/), and quantization let \&AI optimize deeply. Their AI patent agent, Andy, uses natural language to guide attorneys through patent analysis tasks, drastically cutting time-to-result.
-*"With Qdrant, we scaled to a billion vectors and still respond in sub-second latency. That lets us power workflows that used to take hours in just a few minutes."*
-
## Unlocking new markets and workflows
\&AI’s ability to search across the global patent corpus opened doors to new jurisdictions and legal use cases. It also gave them the confidence to offer strong guarantees to clients: yes, we’re looking at *everything*.
@@ -86,7 +116,11 @@ Their semantic-first retrieval engine also enabled new products, like real-time
\&AI is already working on the next version of Andy, expanding natural language capabilities and increasing automation in patent workflows. With Qdrant's upcoming inference capabilities and support for hybrid and multimodal search, Turner sees room for deeper integration.
-"We want to stay at the application layer. If Qdrant can keep lifting the infrastructure complexity off our plate, we’re happy to keep building on it."
+{{< quote
+ text="We want to stay at the application layer. If Qdrant can keep lifting the infrastructure complexity off our plate, we’re happy to keep building on it."
+ name="Herbie Turner"
+ role="CTO & Co-Founder"
+ company="&AI" >}}
As legal AI matures, \&AI’s retrieval-first approach — and Qdrant’s infrastructure support — are helping bring clarity and trust to one of the most high-stakes domains in AI.
diff --git a/qdrant-landing/content/blog/case-study-bayer.md b/qdrant-landing/content/blog/case-study-bayer.md
new file mode 100644
index 000000000..3781580a8
--- /dev/null
+++ b/qdrant-landing/content/blog/case-study-bayer.md
@@ -0,0 +1,251 @@
+---
+title: "How Bayer Built an Enterprise-Scale Search Engine with Qdrant"
+draft: false
+slug: case-study-bayer
+short_description: "Bayer serves 116,000 employees and grounds deep agents on a Qdrant Hybrid Cloud deployment."
+description: "How Bayer built myGenAssist on Qdrant Hybrid Cloud: 135M points, hybrid search for deep agents, semantic caching, multitenancy, and a 20% efficiency gain."
+preview_image: /blog/case-study-bayer/social_preview.png
+social_preview_image: /blog/case-study-bayer/social_preview.png
+date: 2026-08-13T00:00:00.000Z
+author: Daniel Azoulai
+featured: false
+tags:
+ - Bayer
+ - case study
+ - vector search
+ - hybrid search
+ - hybrid cloud
+ - agentic AI
+ - enterprise search
+ - life sciences
+partition: case-studies
+---
+
+
+
+Bayer is a global life sciences company operating at the intersection of two of the most consequential fields in human life: health and nutrition. Its pharmaceutical work supports drug discovery and patient care, while its crop science work supports food production at planetary scale. The company's guiding ambition, "Health for all, hunger for none," frames how it thinks about technology: AI is not a side project, but a lever applied across the entire organization, from improving the productivity of colleagues to accelerating yield prediction and drug discovery.
+
+Turning that ambition into production systems for 116,000 employees is a hard infrastructure problem. It requires retrieval that stays fast under sustained load, grounds large language models (LLMs) in real data to suppress hallucinations, satisfies strict life sciences compliance requirements, and adapts as the underlying AI workloads shift from simple chatbots to autonomous agents. This is the story of how Bayer built that foundation, and why Qdrant has sat at the center of it for nearly three years.
+
+{{< quote
+ text="People used to look at vector databases only for RAG applications. Now they're solving enterprise search problems. No one had an omnimodal search engine where you could search videos, audio, and every asset the company generates. We realized Qdrant was turning into that."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer"
+ avatar="/img/customers/hooman-sedghamiz.svg"
+ logo="/img/brands/bayer.svg"
+ featured="true" >}}
+
+## A Platform Born Weeks After ChatGPT
+
+Hooman Sedghamiz has spent roughly 15 years applying AI across healthcare, from medical devices to drug discovery research. For most of that time, AI was a tool for specialists running research projects. At Bayer, that changed once ChatGPT's chat interface made AI useful to almost every employee.
+
+Bayer moved quickly. Within three months, Sedghamiz's team stood up myGenAssist, an internal generative AI platform. At the time, the vector search landscape was small: only a handful of companies offered it. myGenAssist started with a thousand users and a focused set of natural language processing applications for drug discovery, drawing on data sources such as the FDA and PubMed.
+
+From there, it scaled into a full platform layer. Today, myGenAssist serves the entire company, processes over 1.5 million messages a month, and has ingested more than 450,000 uploaded documents, all while keeping the complexity of RAG hidden from the end user.
+
+{{< quote
+ text="The whole stack of RAG is hidden from users. For them it's just a file upload, but it ends up going through several layers of retrieval-augmented generation, Qdrant being part of it. That has proven to be quite successful to bring grounding and reduce hallucinations for LLM applications."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+## Choosing a Vector Search Engine, Three Years Ago
+
+Three years ago, Bayer evaluated vector search. The company's first prototype, MVP1, ran on Redis. But Redis was not scalable enough for what Bayer needed, so the team benchmarked across other providers, including Qdrant.
+
+The team evaluated across several dimensions: price-performance ratio, latency, and openness. Qdrant was open source, which meant the team could test it fast without procurement friction. Latency was strong. And it was written in Rust, a signal of the memory efficiency and predictable performance that life sciences workloads would later demand.
+
+{{< quote
+ text="There weren't many options back then. We did benchmarking across price-performance, latency, and other aspects. The first points we really liked: it was open source, we could test it very fast, latency was good, and it was written in Rust. We ended up going with Qdrant."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+## From Self-Hosted to Hybrid Cloud: Meeting Compliance Without Drowning in Ops
+
+Bayer began with self-hosted Qdrant. That worked at first, but as the platform scaled from a thousand users toward the full company of 116,000, the operational burden grew. Strict requirements made the picture more complex. As a life sciences company, Bayer needs systems running in its own certified cloud, with data that does not leave its premises.
+
+Pure self-hosting satisfied the compliance side but became demanding for a team that, in Sedghamiz's words, is not large. The answer was [Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/), using the Kubernetes operator. The arrangement keeps data inside Bayer's environment to meet its compliance posture, while offloading the heavy lifting of cluster management. Bayer has run this hybrid model for more than two years.
+
+
+
+{{< quote
+ text="Life science companies want data to stay inside and use the platform self-hosted if possible. But self-hosting was already quite demanding for us. Our team is not that big, so we decided to use hybrid management."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+## Scaling to Millions of Messages and an Evolving Data Model
+
+The scale of the deployment is substantial. Behind it sits a four-node Qdrant Hybrid Cloud cluster holding roughly 135 million points across seven collections. The largest single collection, the user file store, holds 91 million points as dense plus sparse hybrid vectors.
+
+Much of the operational strain has come not from Qdrant itself but from the surrounding pipeline. Users treat the platform like a file drive, re-uploading and revising the same documents, which means vectors must stay continuously synced with the source documents. Document parsing, which runs before vectorization, is heavy and has been a recurring source of load. Keeping the vector store backfilled and consistent as documents change has been one of the central engineering challenges.
+
+The data model itself is also expanding. Bayer is migrating toward an omnimodal approach, driven by the reality of life sciences data: medical images, X-rays, CT scans, and molecular databases sit alongside text. With embedding models that handle multiple modalities, the team can now treat search as a problem across all enterprise assets, not just documents.
+
+{{< quote
+ text="For every enterprise, it's very important to be able to find assets no matter what format they're in: images, text, a molecular image, anything. We've started looking at these not just for simple RAG applications, but to let people search across all the assets they're dealing with."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+The omnimodal pipeline is concrete, not aspirational. When a scientific PDF enters the system, a vision model generates search-optimized descriptions of every figure (content summary, OCR'd labels, key concepts) and injects them into the text stream before chunking.
+
+So searching "receptor binding affinity curve" returns the figure itself, not just paragraphs that mention it. Audio and video recordings are chunked, transcribed via Whisper, and vector-indexed alongside text documents. A lab meeting from three months ago becomes searchable in Qdrant within minutes of upload.
+
+## What Users Actually Care About: Latency and Grounded Answers
+
+For the people querying the platform, two things matter most. The first is latency. Users expect responses in under 10 seconds, so a research query that takes longer to return a simple answer erodes the experience. Fast retrieval is a core part of that budget, and Qdrant is a major component of it.
+
+The second is retrieval quality. Grounded, high-quality answers are what keep users satisfied and drive measurable productivity gains. Hallucinations do the opposite. This is where [hybrid search](https://qdrant.tech/documentation/concepts/hybrid-queries/) became decisive. Two years ago, semantic search alone was not enough. The combination of keyword and semantic retrieval in a single query proved far more capable, and it is now central to how Bayer's agents find relevant context.
+
+{{< quote
+ text="You want your retrieval to be very fast. A low-latency platform helps the user experience a lot. And the second point is the quality of retrieval inside that latency. It's important that your vector search supports hybrid search, for example. That's great."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+
+
+## Composable Retrieval for Agents: Exposing Qdrant Directly to the Model
+
+The most significant shift in Bayer's architecture is the move from chatbot-style interactions to deep agents. Bayer now runs agentic applications on a LangGraph-based harness, comparable to the deep research and coding agents that have become common, and these agents are far hungrier for search than the simpler systems that preceded them. A single deep research run unrolls a long tool-calling loop, hundreds of steps deep, and can fire thousands of retrieval queries before it returns, which makes per-query latency matter even more than it did before. Every one of those tools, from web search to PubMed to the FDA connector, is itself backed by a Qdrant collection.
+
+{{< quote
+ text="If you don't have a Qdrant vector store behind the scenes, it's very difficult to ground LLMs into reality. If you remove the search from the agent, the results go back to two years ago."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer"
+ avatar="/img/customers/hooman-sedghamiz.svg"
+ logo="/img/brands/bayer.svg" >}}
+
+Crucially, Bayer exposes the Qdrant API directly to the model. Whether the requester is a human or an agent, the same interface is available, and the agent can choose how to retrieve based on the task. This is exactly the composable model Qdrant is designed for: retrieval primitives the caller combines at query time, rather than a fixed pipeline hidden behind an opaque API. Under the hood, the hybrid path runs dense and sparse queries in parallel and fuses them with Reciprocal Rank Fusion before a BGE reranker sharpens the final ordering. The agent sees a clean set of tools, not that machinery.
+
+
+
+A feature Bayer calls knowledge bases makes this concrete. Users start a project, drop in folders of data in any format, and the agent works against that data much like a coding agent works against a file system. Bayer extended the agent's command set so that when keyword search fails, it can escalate to semantic or hybrid search through the Qdrant API. It can also fan a single question into several reformulations (keywords, a question form, a hypothetical answer) and search them at once. The agent decides which retrieval strategy fits the moment.
+
+{{< quote
+ text="This is very powerful because the agent now decides: I didn't find anything with keyword search, so I can switch to semantic search, or I can use hybrid search that the API exposes to me. Two years ago, semantic search alone wasn't enough. Now with hybrid search it's way more powerful."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+This is also where the omnimodal direction pays off. Users upload meeting transcripts, images, audio, and video, and the agent discovers and connects them. Qdrant increasingly serves as the agent's memory, letting it recall what a user has been working on and tailor answers accordingly. Teams elsewhere in the company can point their own applications at a shared collection to build their own search experiences, from molecule search to internal enterprise search.
+
+Retrieval quality benefits from a parallel query expansion strategy. A single user question generates four Qdrant searches simultaneously: the verbatim query, a question-form rewrite, extracted keywords, and a HYDE hypothetical answer. Results are fused, deduplicated with a diversity cap of three chunks per document, and optionally reranked. The approach is particularly effective for pharmaceutical literature, where the same concept appears under different nomenclatures across regulatory filings, clinical protocols, and marketing materials.
+
+Document parsing itself is agentic. Rather than pre-processing every upload through expensive OCR, the platform defers extraction until the agent actually needs a document's content. A lightweight sandbox-local parser handles simple formats instantly; complex PDFs with tables and figures fall through to server-side Docling OCR on demand. Results are cached and indexed into Qdrant on first use. With 450,000 documents uploaded and most never read beyond their metadata, this lazy parsing strategy cuts compute costs by roughly 80 percent compared to eager processing. This agentic parsing pipeline (the tiered extraction, the lazy on-demand OCR, and the path that turns a raw upload into Qdrant-indexed content the moment an agent reaches for it) was built by Balkrushn Hirani, myGenAssist's backend developer lead.
+
+The filesystem metaphor runs deeper than an API wrapper. myGenAssist mounts each knowledge base as a virtual directory, `/kb/{id}/`, and exposes standard Unix operations: `ls`, `read_file`, `grep`, `glob`. The critical innovation is that `grep semantic:drug interaction` transparently dispatches to Qdrant hybrid search. The agent decides at runtime whether a literal grep or a semantic search will answer the question better, switching strategies mid-task without human intervention. Researchers interact with their document collections the way a developer interacts with a codebase. Much of this retrieval architecture (the knowledge base backend, the chunking strategy that decides how documents are split and embedded, and the semantic-grep dispatch into Qdrant) is the work of Wiktor Sobanski, one of myGenAssist's senior backend engineers, who owns how documents move from raw upload to searchable vector.
+
+## The AI Hub: One Search Fabric for People and Agents
+
+The composable philosophy does not stop at documents. As the platform grew, the assets worth finding were no longer just files. They were the things people built on top of myGenAssist: assistants, tools, MCP servers and their tools, workflows, knowledge bases, skills, and artifacts.
+
+Bayer unified all of them into a single searchable catalog it calls the AI Hub, where more than 83,000 of these reusable building blocks now live behind one search box: roughly 28,000 assistants, 22,000 artifacts, 20,000 knowledge bases, and close to 10,000 workflows among them.
+
+The Hub runs the same hybrid search playbook Bayer proved on its vector infrastructure: every solution carries a 1024-dimension BGE-M3 embedding, and a single search function fuses dense vector similarity with full-text keyword matching using Reciprocal Rank Fusion. The fused results are then re-ranked by live quality signals (popularity, star ratings, reliability, and recency) so the best-loved, most reliable tools rise to the top. Employees lean on it hard: the Hub serves roughly 49,000 searches a month across more than 7,500 distinct people.
+
+
+
+And agents use the exact same search to equip themselves. When a session exposes more than a handful of tools, a tool-discovery layer embeds the user's request, searches the Hub, and hands the model only the dozen or so tools that fit the step. The agent can call a `discover_tools` command to pull in more on the fly, or find a specialist assistant and delegate to it as a subagent.
+
+That same search decides which of a 430-tool surface to build eagerly versus defer, cutting agent startup from more than five seconds to under two. A personal recommendations feed closes the loop, surfacing solutions a given user has not found yet. Build something once, and it becomes discoverable everywhere, by every colleague and every agent on the platform.
+
+The scale of the tool ecosystem creates its own retrieval problem. With more than 100 enterprise tools available, from FDA databases to chemistry engines to internal ServiceNow connectors, sending all tool definitions in every LLM call would consume most of the context window.
+
+Instead, a discovery middleware runs the user's message through AI Hub's Qdrant-backed search on every turn, surfaces only the 12 most relevant tools, and maintains a sticky memory of previously discovered tools via LangGraph checkpoints. The effect is a 70 percent reduction in prompt tokens while keeping every tool reachable.
+
+## Caching Search to Control Agent Cost
+
+Deep agents do not just stress latency; they stress cost. Bayer's agents sometimes run thousands of online search queries, and repeatedly calling external APIs for the same large articles is expensive. To control this, the agent's web-scraping connector writes the chunks it fetches straight back into a dedicated Qdrant collection. The next time a similar question comes in, the agent checks the vector store for something close to what it read before, rather than paying to call the external API again.
+
+
+
+This turns Qdrant into a semantic cache layer for agentic workloads, reducing both cost and latency on repeated queries. It also surfaces a hard open problem: keeping cached content fresh when the underlying source changes. If an article read last week is edited the week after, the cached version drifts. Managing that synchronization, alongside improving recall, is an ongoing area of work. The semantic cache layer is part of that same body of backend work led by Sobanski, who has focused on keeping agent retrieval both cheap and fresh as the platform scaled.
+
+Qdrant also powers the agent's persistent memory. A dedicated collection stores user-assistant message pairs as hybrid vectors, scoped by account and assistant identity. When a user returns days or weeks later, the agent proactively searches this collection using Reciprocal Rank Fusion with temporal weighting: recent interactions rank higher, but nothing is forgotten. A daily background job backfills any gaps. Now, a researcher can say "continue the analysis we started last Tuesday" and the agent picks up exactly where it left off, grounded in the actual prior exchange rather than a summary.
+
+## Scaling Retrieval: Lessons From the Road to 116K Employees
+
+When myGenAssist served a thousand users, a single Qdrant collection with default settings was sufficient. Documents went in, vectors came out, and search simply worked. Scaling to 116,000 employees meant moving from an experimental prototype to a full-scale production system, one that tests the absolute limits of the surrounding architecture.
+
+Instead of a hard migration, the team adopted a dual-collection strategy. They stood up the new collection alongside the legacy one and routed all new uploads there. At query time, the retriever fans out to both collections, merges the results, and deduplicates them. The legacy collection remains unaware of the new embedder, and the new collection knows nothing of the legacy documents. Over time, as data retention policies deleted older files, the legacy collection naturally aged out and was eventually shut down completely. This meant no expensive recomputation of historical vectors was ever needed: the only overhead was a single additional embedding call for the user's query, a marginal cost for a seamless, zero-downtime migration.
+
+The team's indexing strategy also evolved to match access patterns. For knowledge base collections, the global HNSW index is disabled entirely (`m=0`). Because every query is scoped to a single tenant, building a global graph connecting documents that will never be searched together is wasted compute. Instead, Qdrant relies on payload indexes to build per-tenant subgraphs, ensuring one team's 50,000 documents do not slow down another team's 500. Conversely, curated global datasets like PubMed retain the global index because they are searched without tenant filters.
+
+To manage the memory footprint of this growing dataset, Bayer relies on binary quantization. Quantized vectors remain in RAM for fast initial retrieval, while the full-precision originals are kept on disk. Searches hit the compact index first, then rescore the top candidates against the originals using 3x oversampling to maintain high recall. This architecture allows the cluster to fit within memory limits without doubling infrastructure costs.
+
+
+
+These solutions were not about chasing the latest algorithmic trends; they were the practical, battle-tested realities of scaling a system for an enterprise workforce. By solving for state, consistency, and memory at scale, Bayer transitioned from a promising experiment to a hardened, enterprise-grade retrieval engine, one capable of supporting the company's shift toward autonomous agents and delivering measurable business impact.
+
+The collection schema itself reflects enterprise realities. Named vector spaces, dense and sparse, coexist in a single collection. A full-text payload index with word tokenization enables MatchText filtering for exact regulatory identifiers. And Qdrant's native `is_tenant` flag on `knowledge_base_id` partitions query execution so that a single shared collection serves more than 10,000 knowledge bases with per-tenant isolation. No cross-contamination, no per-tenant infrastructure overhead.
+
+Operational resilience at this scale demands coordination across pods. A distributed circuit breaker, implemented with Redis Lua atomics for state transitions, protects all Qdrant operations. If indexing workers detect latency spikes, API pods fail fast within milliseconds rather than queuing requests behind a stalled connection. The circuit's half-open probing ensures automatic recovery without human intervention.
+
+## Measured Outcomes: 20% Efficiency, and a Moving Target
+
+Bayer measures impact through KPI surveys run every six months across two user groups: general users seeking time savings, and researchers running deeper, higher-budget agentic workflows. The headline result so far is a roughly 20% efficiency gain from using the AI platform.
+
+The team is careful not to over-attribute. It does not isolate which component drives which fraction of the gain. But the connection to retrieval is direct: a large part of the efficiency comes from getting grounded responses, and grounding is impossible without the vector store underneath. Hallucinated answers tank survey scores; grounded answers lift them.
+
+{{< quote
+ text="We've seen 20% efficiency when it comes to using AI platforms. A big part of that gain is that you have to get results from AI that are grounded. If you get hallucinations, people are not satisfied. It wouldn't be possible without the components."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+With the recent shift to more autonomous agents, the measurement problem itself is evolving. Earlier chatbot-style systems delivered incremental time savings: a faster email summary, a quicker draft. The new agents can run for 10 to 15 minutes unattended and return a completed task: research done, document written, a notification sent to the user's phone. That changes the question from "how much time did we save" to "how well was the whole task done," a harder thing to quantify but a larger prize.
+
+That last mile, meeting people where they already work, runs through myGenAssist Claw, the Microsoft Teams integration built by Hendrik Hogertz (myGenAssist senior developer), which lets an employee @-mention the assistant inside a Teams channel and hand it a task without ever leaving the conversation.
+
+But efficiency is a means, not an end. The real question for a company like Bayer is whether the platform can accelerate what the company exists to do.
+
+In 2026, the ambition moves beyond efficiency. The platform is orienting toward the core missions of a life sciences company: agentic drug discovery pipelines where autonomous agents navigate literature, chemical databases, and clinical evidence to surface novel hypotheses; chemical research workflows where agents propose, evaluate, and iterate on molecular candidates with human scientists in the loop; and regulatory preparation where agents assemble submission packages from scattered internal knowledge, every claim grounded and every citation verified. The retrieval layer, Qdrant, becomes the connective tissue that makes these workflows possible, because an agent that cannot find the right paper, the right structure, or the right prior result at the right moment cannot do science.
+
+The collaboration model itself is bidirectional and auditable. When an agent produces an artifact (a research report, a data visualization, a presentation, or a structured analysis), it renders live in the user interface. The scientist can edit it directly: refine a conclusion, correct a chemical structure, adjust a figure. The agent sees those edits on the next turn and incorporates them, creating a transparent co-authoring loop between human expertise and machine scale. Every step of this exchange, every retrieval, every generation, every human edit, is traced in Langfuse with full cost attribution and latency breakdowns. For scientific discovery, where reproducibility and audit trails are non-negotiable, this means any result can be reconstructed: which sources were consulted, which model produced the synthesis, and where the human refined the output.
+
+In a regulated industry, trust is not optional. Every retrieval hit from Qdrant surfaces as an inline citation in the user interface: a clickable reference that reveals the exact chunk text, source document, and a deep link into the knowledge base viewer. Users can inspect and even edit the source data without leaving the conversation. For pharmaceutical compliance, where every claim must trace back to an authoritative source, this closes the loop between AI-generated answers and auditable evidence.
+
+Traceability extends beyond user-facing citations into the infrastructure itself. Every agent session, from tool selection through retrieval to final response, is captured as a structured trace in Langfuse, with cost attribution per model call and latency breakdowns per middleware hop. Prometheus metrics track Qdrant operation health in real time: query latency percentiles, circuit breaker state transitions, RRF fallback rates, and collection-level indexing throughput. For a life sciences company operating under GxP expectations, this observability layer is not a luxury. It means that when a regulatory auditor asks how a particular answer was generated, the platform can reconstruct the full retrieval path, which collections were queried, which chunks scored highest, and which model produced the synthesis, down to the millisecond.
+
+## Staying Current: Quantization, Indexing, and Feature Velocity
+
+The Bayer team actively tracks Qdrant releases and adopts performance features as they ship. It uses incremental HNSW indexing to absorb the constant stream of document updates without full reindexing. It adopted binary quantization to compact points and reduce the memory footprint shortly after release. One engineer recently went through the latest Qdrant publications to update the team's search strategy and apply current optimizations.
+
+{{< quote
+ text="Qdrant is one of the more feature-rich platforms where you can do all those things directly inside the vector store. We always try to be on top of the features you push out to reduce the memory footprint and the latency."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+This matters to Bayer because it reduces the gap between a published optimization and a deployed one. When Qdrant ships something like improved compression, Bayer can fold it into a live, compliance-bound, enterprise-scale platform without re-architecting.
+
+## Why a Composable Engine, Not a Black Box
+
+Bayer's experience drove an architectural conviction to focus on retrieval rather than agent orchestration. Even when Bayer ships an end-to-end agent that handles everything, its users still prefer access to the underlying pieces. Developers building on the platform's API want lower-level components they can inspect and optimize, not an opaque pipeline they have to trust blindly.
+
+{{< quote
+ text="People still prefer to have access to these pieces themselves, like Qdrant. It's very important to give developers a platform that's composable, where they can optimize each part and build their own workflows. Not all agentic pipelines are applicable to all use cases."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
+
+That preference is sharpened by the proliferation of hyperscaler agent frameworks. With Google, AWS, and Azure each pushing their own solutions, teams struggle to manage and optimize systems they cannot see into. A composable engine that exposes its retrieval primitives lets engineers build pipelines tuned to their specific workload, rather than accepting opaque defaults.
+
+## What's Next
+
+Bayer's roadmap continues to push on the dimensions that drew it to Qdrant in the first place: lower and more predictable latency, higher retrieval quality, and broader deployment options. The team plans to scale its clusters further as data and application count grow, expand its omnimodal search capabilities, and deepen the observability and regression testing around its retrieval pipeline.
+
+## From Prototype to Enterprise Search Engine
+
+Bayer started with a Redis prototype and a thousand users. Three years later, it runs a compliance-bound, Hybrid Cloud deployment serving 116,000 employees, processing millions of messages a month, grounding autonomous agents, and increasingly searching across every modality the company produces. Qdrant has been the constant underneath that evolution: the retrieval layer that keeps answers grounded, the API the agents call directly, and the composable foundation that has adapted as Bayer's AI workloads shifted from chatbots to agents.
+
+{{< quote
+ text="We're turning into the AI search engine for the company. There are various applications for a vector database even beyond simple RAG, beyond the chatbot. It supports memory for the agent, it powers enterprise search, and it lets any team build their own multimodal search engine on top."
+ name="Hooman Sedghamiz"
+ role="Senior Director AI/ML - Precision Medicine & Insights"
+ company="Bayer" >}}
diff --git a/qdrant-landing/content/blog/case-study-dust-v2.md b/qdrant-landing/content/blog/case-study-dust-v2.md
index 0fd9f726f..6e0158396 100644
--- a/qdrant-landing/content/blog/case-study-dust-v2.md
+++ b/qdrant-landing/content/blog/case-study-dust-v2.md
@@ -22,6 +22,8 @@ partition: case-studies

+We first wrote about Dust in 2024, in [Dust and Qdrant: Using AI to Unlock Company Knowledge and Drive Employee Productivity](/blog/dust-and-qdrant/). This is what came next.
+
### The Challenge: Scaling AI Infrastructure for Thousands of Data Sources
Dust, an OS for AI-native companies enabling users to build AI agents powered by actions and company knowledge, faced a set of growing technical hurdles as it scaled its operations. The company's core product enables users to give AI agents secure access to internal and external data resources, enabling enhanced workflows and faster access to information. However, this mission hit bottlenecks when their infrastructure began to strain under the weight of thousands of data sources and increasingly demanding user queries.
diff --git a/qdrant-landing/content/blog/case-study-dust.md b/qdrant-landing/content/blog/case-study-dust.md
index f297f4163..b8767bb60 100644
--- a/qdrant-landing/content/blog/case-study-dust.md
+++ b/qdrant-landing/content/blog/case-study-dust.md
@@ -14,6 +14,8 @@ tags:
weight: 0
---
+*This is Dust’s story as it stood in 2024. For how they scaled further, read [How Dust Scaled to 5,000+ Data Sources with Qdrant](/blog/case-study-dust-v2/).*
+
One of the major promises of artificial intelligence is its potential to
accelerate efficiency and productivity within businesses, empowering employees
and teams in their daily tasks. The French company [Dust](https://dust.tt/), co-founded by former
@@ -62,8 +64,15 @@ strategy with the embeddings models and performs retrieval augmented generation.
For this, Dust required a vector database and evaluated different options
including Pinecone and Weaviate, but ultimately decided on Qdrant as the
-solution of choice. “We particularly liked Qdrant because it is open-source,
-written in Rust, and it has a well-designed API,” Polu says. For example, Dust
+solution of choice.
+
+{{< quote
+ text="We particularly liked Qdrant because it is open-source, written in Rust, and it has a well-designed API."
+ name="Stanislas Polu"
+ role="Co-Founder"
+ company="Dust" >}}
+
+ For example, Dust
was looking for high control and visibility in the context of their rapidly
scaling demand, which made the fact that Qdrant is open-source a key driver for
selecting Qdrant. Also, Dust's existing system which is interfacing with Qdrant,
@@ -90,26 +99,26 @@ more effectively. “This allowed us to scale smoothly from there,” Polu says.
## Results
-Dust has seen success in using Qdrant as their vector database of choice, as Polu
-acknowledges: “Qdrant’s ability to handle large-scale models and the flexibility
-it offers in terms of data management has been crucial for us. The observability
-features, such as historical graphs of RAM, Disk, and CPU, provided by Qdrant are
-also particularly useful, allowing us to plan our scaling strategy effectively.”
+Dust has seen success in using Qdrant as their vector database of choice.
-
+{{< quote
+ text="Qdrant’s ability to handle large-scale models and the flexibility it offers in terms of data management has been crucial for us. The observability features, such as historical graphs of RAM, Disk, and CPU, provided by Qdrant are also particularly useful, allowing us to plan our scaling strategy effectively."
+ name="Stanislas Polu"
+ role="Co-Founder"
+ company="Dust" >}}
Dust was able to scale its application with Qdrant while maintaining low latency
across hundreds of thousands of collections with retrieval only taking
milliseconds, as well as maintaining high accuracy. Additionally, Polu highlights
-the efficiency gains Dust was able to unlock with Qdrant: "We were able to reduce the footprint of vectors in memory, which led to a significant cost reduction as
-we don’t have to run lots of nodes in parallel. While being memory-bound, we were
-able to push the same instances further with the help of quantization. While you
-get pressure on MMAP in this case you maintain very good performance even if the
-RAM is fully used. With this we were able to reduce our cost by 2x."
+the efficiency gains Dust was able to unlock with Qdrant.
+
+{{< quote
+ text="We were able to reduce the footprint of vectors in memory, which led to a significant cost reduction as we don’t have to run lots of nodes in parallel. While being memory-bound, we were able to push the same instances further with the help of quantization. While you get pressure on MMAP in this case you maintain very good performance even if the RAM is fully used. With this we were able to **reduce our cost by 2x**."
+ name="Stanislas Polu"
+ role="Co-Founder"
+ company="Dust"
+ avatar="/img/customers/stanislas-polu.svg"
+ logo="/img/customers-case-studies-logo/dust.svg" >}}
@@ -123,3 +132,5 @@ Dust will expand on its structured data capabilities.
To learn more about how Dust uses Qdrant to help employees in their day to day
tasks, check out our [Vector Space Talk](https://www.youtube.com/watch?v=toIgkJuysQ4) featuring Stanislas Polu, Co-Founder of Dust.
+
+
diff --git a/qdrant-landing/content/blog/case-study-minima.md b/qdrant-landing/content/blog/case-study-minima.md
new file mode 100644
index 000000000..c5de32ba4
--- /dev/null
+++ b/qdrant-landing/content/blog/case-study-minima.md
@@ -0,0 +1,128 @@
+---
+draft: false
+title: "Qdrant and Minima Deliver 2.92x More Agentic RAG Tasks per GPU-Hour"
+short_description: "A joint benchmark of Qdrant retrieval and Minima-optimized inference with Qwen3.6-27B on a single RTX PRO 6000 Blackwell GPU."
+description: "Hybrid search, payload filters, and late-interaction reranking in Qdrant plus Minima-optimized inference delivered 2.92x more successful agentic RAG tasks per GPU-hour without reducing grounded quality."
+preview_image: /blog/case-study-minima/social_preview.png
+social_preview_image: /blog/case-study-minima/social_preview.png
+date: 2026-08-13T00:00:00+00:00
+author: Qdrant and Minima Engineering
+featured: false
+tags:
+ - case study
+ - agentic ai
+ - hybrid search
+ - reranking
+ - inference optimization
+ - benchmark
+---
+
+
+
+## Reducing Retrieval and Calls
+
+When a retrieval-augmented generation (RAG) agent runs, it often has to plan a search, check the evidence it gets back, and try again when that evidence falls short. Those inefficiencies compound. Every extra retrieval and every extra model call adds latency, context, and inference cost.
+
+To attack that cost, Minima built a bounded retrieval agent. It planned the query, searched Qdrant, decided whether the evidence was sufficient, and rephrased the query when it was not. It then generated a cited answer with Qwen3.6-27B. Qdrant handled [hybrid search](https://qdrant.tech/documentation/search/hybrid-queries/), applied [payload filters](https://qdrant.tech/documentation/search/filtering/), and ran [late-interaction reranking](https://qdrant.tech/documentation/tutorials-basics/reranking-hybrid-search/). Minima served each request on a single 96 GB NVIDIA RTX PRO 6000 Blackwell GPU.
+
+Across 1,800 evaluated tasks and 10,000 full agent episodes, the joint stack reached 3,750 tasks per GPU-hour, compared to 1,350 for dense retrieval with BF16 inference. Median task latency fell from 21.3 seconds to 7.7 seconds, grounded task success rose from 80.1% to 84.2%, and successful throughput climbed from 1,081 to 3,158 tasks per GPU-hour.
+
+| First-Pass Evidence | Context per Task | Inference Throughput | Joint Capacity |
+|:---:|:---:|:---:|:---:|
+| 72% to 87% sufficient | 5.2K to 2.3K tokens (56% less) | 392.2 output tokens/s on one GPU | 1,350 to 3,750 tasks/GPU-hour |
+
+## What We Tested
+
+Minima ran 1,800 multi-step tasks across three public benchmarks, SciFact, FiQA, and HotpotQA, plus a fourth set that Minima built to test payload filtering. In that fourth set, every chunk carries a tenant ID and a document version in its Qdrant payload, and every query has exactly one correct tenant-and-version slice. Any result returned from outside that slice counts as a violation, so the set scores pass or fail with no grader judgment involved. This is the set behind the 50,000-query tenant-policy test reported later in this post. Every run used the same agent prompt, tool schema, stopping rule, and a maximum of two Qdrant calls per episode. Answers were capped at 256 output tokens. Minima then replayed 10,000 complete episodes against one million 400-token chunks and ran a separate 50,000-query adversarial filtering test for each retrieval condition.
+
+| Layer | Baseline | Qdrant + Minima |
+|---|---|---|
+| **Agent loop** | Plan, retrieve, check evidence, refine once if needed, answer with citations | Identical prompt, tool schema, stopping rule, and call limit |
+| **Retrieval** | Qdrant dense retrieval, indexed payload filters, top 16 | Dense plus BM25 sparse, RRF fusion, ColBERT-style late-interaction reranking, the same indexed payload filters, top 8 |
+| **LLM inference** | Qwen3.6-27B BF16 weights and BF16 attention KV | Minima NVFP4 W4A4 weights with native Blackwell kernels, FP8 recent and anchor KV, and Minima TQ3 stale KV |
+
+
+
+All three configurations used the same hardware, Qwen3.6-27B checkpoint, sampling settings, agent prompt template, concurrency, and endpoint. Qdrant ran on its own host. Query encoding ran on a separate CPU-only FastEmbed service, on ONNX Runtime, using the same models in all three configurations: `sentence-transformers/all-MiniLM-L6-v2` for 384-dimensional dense vectors with cosine distance, `qdrant/bm25` for sparse vectors, and `answerdotai/answerai-colbert-small-v1` for 96-dimensional late-interaction multivectors. Qdrant stored and searched the sparse vectors and applied its IDF modifier. No GPU did any encoding work, so the RTX PRO 6000 Blackwell was dedicated to the Minima Qwen3.6-27B endpoint. Episode latency and task throughput were measured end to end, including encoding and retrieval, while the GPU-hour and GPU-cost figures count only that dedicated inference GPU. Retrieval strategy and Minima compression were the only variables that changed between runs. Minima accepted a configuration only if grounded task quality stayed within 1 percentage point of BF16 with retrieval fixed, citation quality held, and at least 99.5% of episodes completed. Query vectors were precomputed only for the isolated Qdrant latency measurement. The end-to-end agent run included planning, embedding, retrieval, evidence checking, and generation.
+
+## Why the First Qdrant Call Was Usually Enough
+
+Dense retrieval handled semantic similarity well. Minima added BM25 to recover the names, IDs, and version strings that embeddings can miss. Qdrant fused the two result sets with reciprocal rank fusion (RRF), then reranked the shortlist with token-level late interaction. Payload filters enforced tenant, language, document type, and version at query time.
+
+The numbers below cover retrieval and the agent loop. Recall@10 and nDCG@10 use the same top-10 ranked evaluation list. The agent prompt was then truncated to the stated dense top 16 or hybrid top 8 context budget. "First-pass evidence sufficient" means the agent did not invoke its optional second search. Per-call latency is warmed Qdrant query time with query vectors precomputed.
+
+| Metric | Dense Pipeline | Hybrid plus Reranking | Change |
+|---|:---:|:---:|:---:|
+| Supporting-document recall@10 | 89.6% | 90.2% | +0.6 pp |
+| nDCG@10 | 0.704 | 0.751 | +0.047 |
+| Context precision | 32.8% | 55.7% | +22.9 pp |
+| First-pass evidence sufficient | 72.0% | 87.0% | +15.0 pp |
+| Mean Qdrant calls per task | 1.28 | 1.13 | -11.7% |
+| Retrieved context per task | ~5.2K tokens | ~2.3K tokens | -56.0% |
+| Retrieval latency p50 / p95 | 8.4 / 19.6 ms | 18.7 / 43.2 ms | +10.3 / +23.6 ms |
+| Tenant-policy violations | 0 / 50,000 | 0 / 50,000 | Passed |
+
+Qdrant's p95 query time was 43.2 milliseconds, less than 0.3% of the 20.8-second p95 BF16 agent episode. The first search was sufficient in 87% of tasks, and mean context fell from 5.2K to 2.3K tokens. Even before Minima was enabled, median task latency dropped from 21.3 to 14.6 seconds and successful throughput rose from 1,081 to 1,669 tasks per GPU-hour, a 54% gain.
+
+>"Inside an agent loop, a weak first retrieval costs more than one extra search. It triggers another round of planning, retrieval and inference, so getting sufficient evidence on the first pass is one of the simplest ways to make the whole system faster and more efficient."
+— Sergii Kozyrev, Co-founder and CEO, Minima AI
+
+
+
+## How Minima Accelerated Every Model Call
+
+Minima stored Qwen3.6-27B weights in NVFP4 W4A4 and ran them with native Blackwell kernels. Recent and anchor KV stayed in FP8. Stale pages moved to the 3-bit TQ3 tier. Minima disabled Qdrant vector [quantization](https://qdrant.tech/documentation/guides/quantization/) so the retrieval and LLM inference effects stayed separate.
+
+| Metric | BF16 Reference | Minima | Result |
+|---|:---:|:---:|:---:|
+| Nominal model weights | 54.0 GB | 16.9 GB | 3.20x smaller |
+| Attention KV per active token | 64.0 KiB | 18.3 KiB | 3.50x smaller |
+| Attention KV for one 32K session | 2.00 GiB | 0.57 GiB | 3.50x smaller |
+| Resident 32K sessions before admission failure | 11 | 96 | 8.7x more |
+| Standalone 512-in / 256-out throughput | 206.4 tokens/s | 392.2 tokens/s | 1.90x |
+
+Compression held task quality. With Qdrant retrieval fixed, grounded task success was 84.3% for BF16 and 84.2% for Minima, citation F1 was 90.8% and 90.7%, and valid tool calls were 99.8% for both. The paired task-quality delta was -0.1 percentage point (95% CI [-0.7, +0.5]), which cleared the pre-registered non-inferiority gate.
+
+## The Joint Result: 2.92x More Successful Tasks per GPU
+
+With BF16 unchanged, Qdrant raised raw capacity from 1,350 to 1,980 tasks per GPU-hour. Holding Qdrant fixed, Minima raised it to 3,750. Applying the grounded task success rate gives 3,158 successful tasks per GPU-hour, 2.92x the baseline.
+
+The table below reports agent results at concurrency 8, with final answers capped at 256 tokens. Task rates are wall-clock completions per GPU-hour.
+
+| Configuration | Context | p50 / p95 | Raw Tasks/h | Grounded Success | Successful Tasks/h |
+|---|:---:|:---:|:---:|:---:|:---:|
+| Qdrant dense top 16 + BF16 weights/KV | ~5.2K | 21.3 / 33.8 s | 1,350 | 80.1% | 1,081 |
+| Qdrant hybrid + reranking top 8 + BF16 weights/KV | ~2.3K | 14.6 / 20.8 s | 1,980 | 84.3% | 1,669 |
+| Qdrant hybrid + reranking top 8 + full Minima | ~2.3K | 7.7 / 11.0 s | 3,750 | 84.2% | 3,158 |
+
+*At the rate of USD 1.50 per GPU-hour that Minima used for test accounting, GPU cost per 1,000 successful agent tasks fell from USD 1.39 to USD 0.48, a 65% reduction. This GPU-only comparison excludes the Qdrant host and embedding services. The same provisioned services stayed online across all three conditions, though the hybrid pipeline put more work on them.*
+
+*Minima did not multiply the 3.2x weight compression, 3.5x KV compression, and smaller retrieval context into a single system claim. They affect different bottlenecks. The measured end-to-end results were 2.78x more raw task capacity and 2.92x more successful tasks per GPU-hour.*
+
+## Why This Matters for Agentic RAG
+
+Qdrant and Minima address different costs inside the loop. Qdrant made the first search sufficient more often and reduced the evidence passed to the model on each attempt. Minima reduced the memory and compute cost of planning, checking, and answering.
+
+A production agent is limited by the whole run, not by vector search or model throughput in isolation. In this test, Qdrant improved the evidence passed to the model and Minima increased the amount of inference one GPU could serve. Together they delivered 2.92x more successful tasks without reducing tool-call validity, grounded quality, or citation quality.
+
+## Reproduce This on Your Corpus
+
+If you run agentic RAG on Qdrant, Qdrant and Minima would like to reproduce this benchmark on your corpus and agent loop. [Contact Qdrant](https://qdrant.tech/contact-us/) or [contact Minima](https://mnma.ai) to get started.
+
+## Technical References
+
+The [Qdrant guide to agentic vector search](https://qdrant.tech/articles/agentic-builders-guide/) explains why retrieval latency, memory, filtering, and reranking matter inside multi-step agent workflows.
+
+The [agentic RAG with LangGraph and Qdrant tutorial](https://qdrant.tech/documentation/tutorials-build-essentials/agentic-rag-langgraph/) covers tool selection, repeated retrieval, and stateful agent control flow.
+
+The [Qdrant hybrid and multi-stage queries documentation](https://qdrant.tech/documentation/search/hybrid-queries/) describes dense and sparse prefetch, RRF and DBSF fusion, and multi-stage ranking.
+
+The [Qdrant hybrid search with reranking tutorial](https://qdrant.tech/documentation/tutorials-basics/reranking-hybrid-search/) walks through the dense, sparse, and ColBERT-style late-interaction workflow.
+
+The [Qdrant multivectors and late interaction tutorial](https://qdrant.tech/documentation/tutorials-search-engineering/using-multivector-representations/) covers native multivector representations and MaxSim scoring.
+
+The [Qdrant filtering documentation](https://qdrant.tech/documentation/search/filtering/) describes payload and point-ID conditions for application-defined constraints.
+
+The [NVIDIA RTX PRO 6000 Blackwell product page](https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000/) lists the 96 GB GDDR7 memory and Blackwell FP4 support.
+
+The [Minima site](https://mnma.ai) covers model-weight, KV-cache, and serving optimization.
diff --git a/qdrant-landing/content/blog/case-study-nyris.md b/qdrant-landing/content/blog/case-study-nyris.md
index 34b8c4066..612c7faff 100644
--- a/qdrant-landing/content/blog/case-study-nyris.md
+++ b/qdrant-landing/content/blog/case-study-nyris.md
@@ -58,7 +58,7 @@ As part of their selection process, Nyris evaluated several critical factors to
Nyris has found several aspects of Qdrant particularly beneficial in their production environment:
- **Enhanced Security with JWT**: [JSON Web Tokens](https://qdrant.tech/documentation/security#granular-access-api-keys) provide enhanced security and performance, critical for safeguarding their data.
-- **Seamless Scalability**: Qdrant's ability to [scale effortlessly across nodes](https://qdrant.tech/documentation/distributed_deployment/) ensures consistent high performance, even as Nyris's data volume grows.
+- **Seamless Scalability**: Qdrant's ability to [scale effortlessly across nodes](https://qdrant.tech/documentation/scaling/distributed_deployment/) ensures consistent high performance, even as Nyris's data volume grows.
- **Flexible Search Options**: The availability of both graph-based and brute-force search methods offers Nyris the flexibility to tailor the search approach to specific use case requirements.
- **Versatile Data Handling**: Qdrant imposes almost no restrictions on data types and vector sizes, allowing Nyris to manage diverse and complex datasets effectively.
- **Built with Rust**: The use of [Rust](https://qdrant.tech/articles/why-rust/) ensures superior performance and future-proofing, while its open-source nature allows Nyris to inspect and customize the code as necessary.
diff --git a/qdrant-landing/content/blog/case-study-opentable.md b/qdrant-landing/content/blog/case-study-opentable.md
index 9907d3dd4..b2a8f748f 100644
--- a/qdrant-landing/content/blog/case-study-opentable.md
+++ b/qdrant-landing/content/blog/case-study-opentable.md
@@ -26,7 +26,15 @@ partition: case-studies
When generative AI tools entered the mainstream, OpenTable knew diners would change how they find and choose restaurants. People were beginning to expect conversational, intelligent and context-aware assistants, rather than static search boxes.
-Patrick Lombardo, Staff ML Engineer at OpenTable, recalls that the team wanted to move quickly. “We knew early on that generative AI was going to change user expectations. Concierge was an opportunity for us to transform the way that diners discover restaurants while building the tooling and infrastructure that will support future AI-powered experiences.”
+The team wanted to move quickly.
+
+{{< quote
+ text="We knew early on that generative AI was going to change user expectations. Concierge was an opportunity for us to transform the way that diners discover restaurants while building the tooling and infrastructure that will support future AI-powered experiences."
+ name="Patrick Lombardo"
+ role="Staff ML Engineer"
+ company="OpenTable"
+ logo="/img/customers-case-studies-logo/opentable.svg"
+ featured="true" >}}
That stepping stone is [Concierge](https://www.opentable.com/blog/concierge-ai-dining-assistant/), an AI-powered assistant designed to answer restaurant-related questions in natural language using OpenTable’s data.
@@ -36,7 +44,11 @@ That stepping stone is [Concierge](https://www.opentable.com/blog/concierge-ai-d
For Concierge to succeed, the assistant needed to respond to the vast majority of user questions and every answer had to reflect reality. Incorrect menu items or outdated offerings could erode user and restaurant trust.
-“The primary goal was answerability. We wanted to make sure the model could answer most questions. The second most important was accuracy, so that when the model gave an answer it was correct.” Puyuan Liu, Machine Learning Scientist, OpenTable
+{{< quote
+ text="The primary goal was answerability. We wanted to make sure the model could answer most questions. The second most important was accuracy, so that when the model gave an answer it was correct."
+ name="Puyuan Liu"
+ role="Machine Learning Scientist"
+ company="OpenTable" >}}
Beyond the application logic, the team needed a vector database that could handle sparse embeddings for keyword expansions and fine-grained filtering. Queries often narrowed results to a single restaurant out of more than 60,000, which placed heavy demands on filtering performance.
@@ -50,13 +62,24 @@ Second, Qdrant delivered reliable high-precision filtering. In production, each
Third, Qdrant Cloud provided a deployment path that was simpler than self-hosting.
-Patrick Lombardo summed it up: "Creating a Qdrant Cloud cluster was one of the easiest parts of the project. It just worked."
+{{< quote
+ text="Creating a Qdrant Cloud cluster was one of the easiest parts of the project. It just worked."
+ name="Patrick Lombardo"
+ role="Staff ML Engineer"
+ company="OpenTable"
+ logo="/img/customers-case-studies-logo/opentable.svg" >}}
The production launch was global from the start, allowing Concierge to answer questions about restaurants in many regions without separate deployments.
### Achieving stability and setting the stage for future innovation
-Concierge met its latency target and maintained high answerability without extensive post-launch tuning. Operationally, Qdrant became one of the most stable components in the stack. Ant White, Principal Software Engineer at OpenTable, explained, “Since running it in production, it is a frictionless part of the stack. ”
+Concierge met its latency target and maintained high answerability without extensive post-launch tuning. Operationally, Qdrant became one of the most stable components in the stack.
+
+{{< quote
+ text="Since running it in production, it is a frictionless part of the stack."
+ name="Ant White"
+ role="Principal Software Engineer"
+ company="OpenTable" >}}
### Key takeaways from the Concierge rollout
diff --git a/qdrant-landing/content/blog/case-study-sprinklr.md b/qdrant-landing/content/blog/case-study-sprinklr.md
index a6d4e252a..d2dea44ad 100644
--- a/qdrant-landing/content/blog/case-study-sprinklr.md
+++ b/qdrant-landing/content/blog/case-study-sprinklr.md
@@ -29,7 +29,13 @@ Raghav Sonavane, Associate Director of Machine Learning Engineering at Sprinklr,
*Figure:* Sprinklr’s RAG architecture
-Sprinklr’s platform is composed of four key product suites - Sprinklr Service, Sprinklr Marketing, Sprinklr Social, and Sprinklr Insights. Each suite is embedded with AI-first features such as assist agents, post-call analysis, and real-time analytics, which are crucial for managing large-scale contact center operations. “These AI-driven capabilities, supported by Qdrant’s advanced vector search, enhance Sprinklr’s customer-facing tools such as FAQ bots, transactional bots, conversational services, and product recommendation engines,” says Sonavane.
+Sprinklr’s platform is composed of four key product suites - Sprinklr Service, Sprinklr Marketing, Sprinklr Social, and Sprinklr Insights. Each suite is embedded with AI-first features such as assist agents, post-call analysis, and real-time analytics, which are crucial for managing large-scale contact center operations.
+
+{{< quote
+ text="These AI-driven capabilities, supported by Qdrant’s advanced vector search, enhance Sprinklr’s customer-facing tools such as FAQ bots, transactional bots, conversational services, and product recommendation engines."
+ name="Raghav Sonavane"
+ role="Associate Director of Machine Learning Engineering"
+ company="Sprinklr" >}}
These self-serve applications rely heavily on advanced vector search to analyze and optimize community content and refine knowledge bases, ensuring efficient and relevant responses. For customers requiring further assistance, Sprinklr equips support agents with powerful search capabilities, enabling them to quickly access similar cases and draw from past interactions, enhancing the quality and speed of customer support.
@@ -56,9 +62,23 @@ After evaluating several options of vector DBs, including Pinecone, Weaviate, an
Sprinklr’s transition to Qdrant was carefully managed, starting with 10% of their workloads before gradually scaling up. The transition was seamless, thanks in part to Qdrant’s configurable [Web UI](https://qdrant.tech/documentation/interfaces/web-ui/), which allowed Sprinklr to fully utilize its capabilities within the existing infrastructure.
-“Qdrant’s ability to index [multiple vectors](https://qdrant.tech/documentation/manage-data/vectors/#multivectors) simultaneously and retrieve and re-rank with precision brought significant improvements to our workflow,” Sonavane remarks. This feature reduced the need for repeated retrieval processes, significantly improving efficiency. Additionally, Qdrant’s [quantization](https://qdrant.tech/documentation/manage-data/quantization/) and [memory mapping](https://qdrant.tech/documentation/manage-data/storage/#configuring-memmap-storage) features enabled Sprinklr to reduce RAM usage, leading to substantial cost savings.
+{{< quote
+ text="Qdrant’s ability to index [multiple vectors](https://qdrant.tech/documentation/manage-data/vectors/#multivectors) simultaneously and retrieve and re-rank with precision brought significant improvements to our workflow."
+ name="Raghav Sonavane"
+ role="Associate Director of Machine Learning Engineering"
+ company="Sprinklr" >}}
-Qdrant now plays a key supportive role in enhancing Sprinklr’s vector search capabilities within its AI-driven applications, which is designed to be cloud- and LLM-agnostic. The platform supports various AI-driven tasks, from retrieval and re-ranking to serving advanced customer experiences. “Retrieval is the foundation of all our AI tasks, and Qdrant’s resilience and speed have made it an integral part of our system,” Sonavane emphasizes. Sprinklr operates [Qdrant as a managed service on AWS](https://qdrant.tech/cloud/), ensuring scalability, reliability, and ease of use.
+This feature reduced the need for repeated retrieval processes, significantly improving efficiency. Additionally, Qdrant’s [quantization](https://qdrant.tech/documentation/manage-data/quantization/) and [memory mapping](https://qdrant.tech/documentation/manage-data/storage/#configuring-memmap-storage) features enabled Sprinklr to reduce RAM usage, leading to substantial cost savings.
+
+Qdrant now plays a key supportive role in enhancing Sprinklr’s vector search capabilities within its AI-driven applications, which is designed to be cloud- and LLM-agnostic. The platform supports various AI-driven tasks, from retrieval and re-ranking to serving advanced customer experiences. Sprinklr operates [Qdrant as a managed service on AWS](https://qdrant.tech/cloud/), ensuring scalability, reliability, and ease of use.
+
+{{< quote
+ text="Retrieval is the foundation of all our AI tasks, and Qdrant’s resilience and speed have made it an integral part of our system."
+ name="Raghav Sonavane"
+ role="Associate Director of Machine Learning Engineering"
+ company="Sprinklr"
+ avatar="/img/customers/raghav-sonavane.png"
+ logo="/img/customer-logo/sprinklr.svg" >}}
### Key Outcomes with Qdrant
@@ -111,40 +131,15 @@ Key Observations:

-```json
+```python
data = [
-
-{'system': 'Qdrant', 'index_size': '1,000', 'MAP': 0.98, 'P95 Time': 0.22, 'Mean Time': 0.1, 'QPS': 280,
-
-'Upload Time': 1},
-
-{'system': 'Qdrant', 'index_size': '10,000', 'MAP': 0.99, 'P95 Time': 0.16, 'Mean Time': 0.09, 'QPS': 330,
-
-'Upload Time': 5},
-
-{'system': 'Qdrant', 'index_size': '100,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.23, 'QPS': 145,
-
-'Upload Time': 100},
-
-{'system': 'Qdrant', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.171, 'Mean Time': 0.162, 'QPS': 596,
-
-'Upload Time': 220},
-
-{'system': 'ElasticSearch', 'index_size': '1,000', 'MAP': 0.99, 'P95 Time': 0.42, 'Mean Time': 0.32, 'QPS': 95,
-
-'Upload Time': 10},
-
-{'system': 'ElasticSearch', 'index_size': '10,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.24, 'QPS': 120,
-
-'Upload Time': 50},
-
-{'system': 'ElasticSearch', 'index_size': '100,000', 'MAP': 0.99, 'P95 Time': 0.48, 'Mean Time': 0.42, 'QPS': 80,
-
-'Upload Time': 1100},
-
-{'system': 'ElasticSearch', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.37, 'Mean Time': 0.236,
-
-'QPS': 348, 'Upload Time': 1150}
-
+ {'system': 'Qdrant', 'index_size': '1,000', 'MAP': 0.98, 'P95 Time': 0.22, 'Mean Time': 0.1, 'QPS': 280, 'Upload Time': 1},
+ {'system': 'Qdrant', 'index_size': '10,000', 'MAP': 0.99, 'P95 Time': 0.16, 'Mean Time': 0.09, 'QPS': 330, 'Upload Time': 5},
+ {'system': 'Qdrant', 'index_size': '100,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.23, 'QPS': 145, 'Upload Time': 100},
+ {'system': 'Qdrant', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.171, 'Mean Time': 0.162, 'QPS': 596, 'Upload Time': 220},
+ {'system': 'ElasticSearch', 'index_size': '1,000', 'MAP': 0.99, 'P95 Time': 0.42, 'Mean Time': 0.32, 'QPS': 95, 'Upload Time': 10},
+ {'system': 'ElasticSearch', 'index_size': '10,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.24, 'QPS': 120, 'Upload Time': 50},
+ {'system': 'ElasticSearch', 'index_size': '100,000', 'MAP': 0.99, 'P95 Time': 0.48, 'Mean Time': 0.42, 'QPS': 80, 'Upload Time': 1100},
+ {'system': 'ElasticSearch', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.37, 'Mean Time': 0.236, 'QPS': 348, 'Upload Time': 1150},
]
```
\ No newline at end of file
diff --git a/qdrant-landing/content/blog/case-study-tripadvisor.md b/qdrant-landing/content/blog/case-study-tripadvisor.md
index 3d26f0efa..0af79c593 100644
--- a/qdrant-landing/content/blog/case-study-tripadvisor.md
+++ b/qdrant-landing/content/blog/case-study-tripadvisor.md
@@ -22,6 +22,16 @@ partition: case-studies

+{{< quote
+ text="Qdrant has been crucial for our transformation. When you're dealing with over a billion plus user-generated, multi-modal pieces of content from hundreds of millions of monthly active users across 21 countries, 11M businesses and all the complex user interactions that come with it, you need a way to bring it all together. Now, we can represent everything from hotel preferences to restaurant choices to user behavior in a unified way. And we’re seeing real business results. Users engaging with our AI-powered features like trip planning are showing 2-3x more revenue."
+ name="Rahul Todkar"
+ name_url="https://www.linkedin.com/in/rahultodkar"
+ role="Head of Data and AI"
+ company="Tripadvisor"
+ avatar="/img/customers/rahul-todkar.svg"
+ logo="/img/brands/tripadvisor.svg"
+ featured="true" >}}
+
Tripadvisor, the world’s largest travel guidance platform, is undergoing a deep transformation. With hundreds of millions of monthly users and over a billion reviews and contributions, it holds one of the richest datasets in the travel industry. And until recently, that data, particularly its unstructured content, had incredible untapped potential. Now, with the rise of generative AI and the adoption of tools like Qdrant’s vector database, Tripadvisor is unlocking its full potential to deliver intelligent, personalized, and high-impact travel experiences.
## Activating Billions of Data Assets
@@ -54,10 +64,6 @@ The team is using Qdrant to build a **user graph**, a multidimensional represent
And unlike traditional databases, Qdrant is built for **real-time, unstructured data**, making it ideal for powering conversational AI, search augmentation, and recommendation engines.
-*“Qdrant has been crucial for our transformation. When you're dealing with over a billion plus user-generated, multi-modal pieces of content from hundreds of millions of monthly active users across 21 countries, 11M businesses and all the complex user interactions that come with it, you need a way to bring it all together. Now, we can represent everything from hotel preferences to restaurant choices to user behavior in a unified way. And we’re seeing real business results. Users engaging with our AI-powered features like trip planning are showing 2-3x more revenue.”*
-
-[*Rahul Todkar*](https://www.linkedin.com/in/rahultodkar) *\- Head of Data and AI*
-
## What’s Next
With Qdrant as a foundational layer, Tripadvisor is only just beginning to tap into the power of its data. The team is already exploring new use cases and looking to deepen its integration of vector search across every stage of the customer journey. And as interest grows in shared learnings and industry best practices, Tripadvisor is also helping shape how other companies apply AI in the real world.
diff --git a/qdrant-landing/content/blog/case-study-voiceflow.md b/qdrant-landing/content/blog/case-study-voiceflow.md
index 9621be288..2acabac54 100644
--- a/qdrant-landing/content/blog/case-study-voiceflow.md
+++ b/qdrant-landing/content/blog/case-study-voiceflow.md
@@ -28,7 +28,7 @@ partition: case-studies
As part of this development, the Voiceflow engineering team was looking for a [vector database](/qdrant-vector-database/) solution to power their RAG setup. They evaluated various vector databases based on several key factors:
-- **Performance**: The ability to [handle the scale](/documentation/distributed_deployment/) required by Voiceflow, supporting hundreds of thousands of projects efficiently.
+- **Performance**: The ability to [handle the scale](/documentation/scaling/distributed_deployment/) required by Voiceflow, supporting hundreds of thousands of projects efficiently.
- **Metadata**: The capability to tag data and chunks and retrieve based on those values, essential for organizing and accessing specific information swiftly.
- **Managed Solution**: The availability of a [managed service](/documentation/cloud/) with automated maintenance, scaling, and security, freeing the team from infrastructure concerns.
diff --git a/qdrant-landing/content/blog/clean-vector-database-collection.md b/qdrant-landing/content/blog/clean-vector-database-collection.md
new file mode 100644
index 000000000..1939a3b5e
--- /dev/null
+++ b/qdrant-landing/content/blog/clean-vector-database-collection.md
@@ -0,0 +1,93 @@
+---
+title: "How to Clean Up a Qdrant Collection"
+draft: false
+slug: clean-vector-database-collection
+short_description: "Learn how duplicates, stale data, and mixed retrieval pipelines distort top-k results as vector collections grow."
+description: "Clean up a vector database collection: control duplicate points, track embedding provenance, remove stale records, and protect top-k search quality."
+preview_image: /blog/clean-vector-database-collection/preview/preview.jpg
+social_preview_image: /blog/clean-vector-database-collection/preview/social_preview.jpg
+title_preview_image: /blog/clean-vector-database-collection/preview/title.jpg
+date: 2026-08-10
+author: Dylan Couzon
+featured: true
+tags:
+ - vector-database
+ - vector-search
+ - collection-maintenance
+ - embeddings
+ - data-quality
+---
+
+Every crawl, retried job, and embedding pipeline change writes points into a vector collection. The stored data keeps moving even when the query code never changes, and the top results move with it.
+
+At first, little looks wrong. Search returns results, and latency stays normal. In our baseline run, the context-relevance score sat at 0.92 out of 1.00 while four answers in ten came back wrong, because duplicate chunks and one outdated record were filling the five results the agent could read.
+
+We reproduced this pattern in a Qdrant and [Future AGI](https://futureagi.com/) webinar using a controlled Pokédex collection. The example was small enough to inspect by hand, but the same failure modes appear in product catalogs, support knowledge bases, recommendations, and agent memory.
+
+## Growth Exposes Flaws That Shipped With the First Ingest
+
+A larger collection creates more competition for each result slot. Weak identity rules, stale records, and poor retrieval choices are usually there from the first ingest. Growth only raises how often a query meets them.
+
+We ran 33 of our 37 test questions against the collection at three sizes, from 1,314 points up to 22,946, and changed nothing else.
+
+{{< figure src="/blog/clean-vector-database-collection/recall-decay.png" alt="Two lines plotted against collection size for the same 33 questions. The share of questions that find the right chunk in the top five falls from 67% at 1.3k points to 39% at 22.9k points, while the share of top-five slots holding repeats rises from 44% to 52%. The two lines cross at around 8.4k points." caption="Recall falls as duplicates take a larger share of the top five." width="100%" >}}
+
+The smallest collection is the one to look at: with 1,314 points, repeats already held 44% of the five slots.
+
+When quality falls after an ingest, inspect the collection before changing prompts or agent logic. Compare the point count with the source count, sample the top-k results your application receives, and check whether the same content or an older version of it appears more than once.
+
+## Deduplication Works by Freeing Result Slots
+
+Repeated ingestion had turned 8,416 distinct points into 22,946 total points. Removing the 14,530 unintended copies dropped queries with a duplicate in the top five from 36 of 37 to 2 of 37, and answer correctness, scored against known-good answers, rose from 0.57 to 0.76.
+
+The agent reads the first five results of each search and nothing below them, so those five slots are all the evidence one search can offer. With the copies gone, they held five different chunks, and answer quality followed. The point count fell as a side effect.
+
+Stable point IDs prevent most duplication at the source. Qdrant point loading is [idempotent](/documentation/manage-data/points/#idempotence), so a retry under the same ID updates the point already there instead of adding another one. Duplicates accumulate when each ingest assigns fresh IDs to content the collection already holds.
+
+An exact copy is easy to find: hash the text of each point and compare the hashes. Deleting one takes more care, because the same text can legitimately sit in two places, once per tenant or once per language. Records that are merely similar have no automatic rule, because whether they count as the same thing depends on what your application does with them.
+
+## Better Retrieval Can Still Produce a Worse Answer
+
+Retrieval quality and answer quality move independently, so we grouped the 37 test questions to exercise one failure at a time. A stronger embedding model took the 14 questions built on near-identical entries from 0.64 to 1.00 on Recall@5, and answer correctness across the full set reached 0.92.
+
+Hybrid search adds two steps to the query path, fusion and reranking. On the 18 ranking questions, fusing sparse and dense results scored 0.72, below the 0.78 plain dense search already reached, and the ColBERT reranking step is what carried the group to 0.89. Fusion on its own would have been a regression.
+
+Then answer correctness fell, 0.92 to 0.86, on the change that improved every ranking metric. The agent had been rewording failed queries and searching again, so the answer column never registered the ranking problem and had nothing to gain from the fix. Searches per question dropped from 2.0 at baseline to 1.14, which is where the improvement showed up instead.
+
+Two checks disagreed about those same answers. Groundedness passed them, because the claims did come from the retrieved text. The hallucination check posted its worst reading of the run, because the answers also carried detail the sources never mentioned.
+
+An agent that retries covers for bad retrieval, which is why the answer column missed both the problem and the fix.
+
+## Freshness Belongs in the Data Model
+
+Similarity can't decide which of two conflicting records is current. An older policy, price, or product state may be a close semantic match and still be wrong for the request.
+
+Our collection contained one outdated type-chart record that was correct for its historical version. Every retrieval and grounding check passed the answer built on it, because the answer reflected that record accurately. Only the correctness judge stayed red, and it stayed red through all four retrieval upgrades.
+
+{{< figure src="/blog/clean-vector-database-collection/steel-stale-baseline.png" alt="The Pokedex app answers the question Does the Steel type resist Ghost and Dark attacks with Yes. The retrieval panel beside it holds the record typechart-steel-gen5 at rank one with a score of 0.639, and the same record again at rank two, tagged as a duplicate." caption="The first question of the session. The top results are copies of an outdated type chart, and the answer built on them is wrong." width="100%" >}}
+
+An `is_current` payload filter removed the stale record from current queries, which recovered answer correctness to 0.92.
+
+{{< figure src="/blog/clean-vector-database-collection/steel-current-filtered.png" alt="The same question answered with No. The retrieval panel is tagged hybrid and current-only, and its top result is the record typechart-steel-gen6, the current type chart." caption="The same question with all four fixes in place. The panel retrieves current records only, and the answer flips." width="100%" >}}
+
+Old records can stay in the collection, as long as something marks which of them is current. `status`, `version`, `is_current`, and `updated_at` are the usual fields, and they let each query state what it should retrieve. A [payload index](/documentation/manage-data/indexing/#payload-index) on every one of them keeps lifecycle rules part of retrieval, and it is what lets [filtering](/documentation/search/filtering/) run at all on a cluster that rejects unindexed fields.
+
+## Pick Metrics That Can See the Failure
+
+Chunk utilization, which measures how much of the retrieved context the generator uses, read 0.85 before deduplication and 0.85 after, while answer correctness rose 0.19 over the same change. Five copies of one chunk score the same as five distinct chunks, so the metric never had a way to register duplication.
+
+One score rarely locates the failing layer, and two read together usually do. Low context relevance with low chunk utilization points at retrieval. High relevance with a failing correctness or groundedness score points at the generator or at the data behind it, and that is the reading that would have pointed at the outdated type chart four stages earlier.
+
+One failure stays invisible to every check in this post: a record that never got ingested. Retrieval metrics score what came back, and answer metrics score what the agent said. Comparing the collection against the source list is the check that catches it.
+
+## Watch the Full Walkthrough
+
+The recording follows these problems through a working RAG system. [Dylan Couzon](https://www.linkedin.com/in/dcouzon) changes the retrieval path in Qdrant, while [Rishav Hada](https://in.linkedin.com/in/rishavhada) traces and evaluates the agent in Future AGI.
+
+
+
+
+
+
+
+[Watch the recording on YouTube](https://www.youtube.com/watch?v=o73V446Po_o), or see [Future AGI's partner recap](https://futureagi.com/blog/why-did-my-rag-agent-get-worse-webinar-2026/?utm_source=10augln&utm_medium=organic&utm_campaign=product_marketing) for more on its evaluation workflow.
diff --git a/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md b/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md
index e7a1d232f..4becace44 100644
--- a/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md
+++ b/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md
@@ -46,7 +46,7 @@ Qdrant is highly scalable and performant: it can handle billions of vectors effi
- **Advanced Similarity Search:** Qdrant supports various similarity [search](https://qdrant.tech/documentation/search/search/) metrics like dot product, cosine similarity, Euclidean distance, and Manhattan distance. You can store additional information along with vectors, known as [payload](https://qdrant.tech/documentation/manage-data/payload/) in Qdrant terminology. A payload is any JSON formatted data.
- **Built Using Rust:** Qdrant is built with Rust, and leverages its performance and efficiency. Rust is famed for its [memory safety](https://arxiv.org/abs/2206.05503) without the overhead of a garbage collector, and rivals C and C++ in speed.
-- **Scaling and Multitenancy**: Qdrant supports both vertical and horizontal scaling and uses the Raft consensus protocol for [distributed deployments](https://qdrant.tech/documentation/distributed_deployment/). Developers can run Qdrant clusters with replicas and shards, and seamlessly scale to handle large datasets. Qdrant also supports [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/) where developers can create single collections and partition them using payload.
+- **Scaling and Multitenancy**: Qdrant supports both vertical and horizontal scaling and uses the Raft consensus protocol for [distributed deployments](https://qdrant.tech/documentation/scaling/distributed_deployment/). Developers can run Qdrant clusters with replicas and shards, and seamlessly scale to handle large datasets. Qdrant also supports [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/) where developers can create single collections and partition them using payload.
- **Payload Indexing and Filtering:** Just as Qdrant allows attaching any JSON payload to vectors, it also supports payload indexing and [filtering](https://qdrant.tech/documentation/search/filtering/) with a wide range of data types and query conditions, including keyword matching, full-text filtering, numerical ranges, nested object filters, and [geo](https://qdrant.tech/documentation/search/filtering/#geo)filtering.
- **Hybrid Search with Sparse Vectors:** Qdrant supports both dense and [sparse vectors](https://qdrant.tech/articles/sparse-vectors/), thereby enabling hybrid search capabilities. Sparse vectors are numerical representations of data where most of the elements are zero. Developers can combine search results from dense and sparse vectors, where sparse vectors ensure that results containing the specific keywords are returned and dense vectors identify semantically similar results.
- **Built-In Vector Quantization:** Qdrant offers three different [quantization](https://qdrant.tech/documentation/manage-data/quantization/) options to developers to optimize resource usage. Scalar quantization balances accuracy, speed, and compression by converting 32-bit floats to 8-bit integers. Binary quantization, the fastest method, significantly reduces memory usage. Product quantization offers the highest compression, and is perfect for memory-constrained scenarios.
diff --git a/qdrant-landing/content/blog/ecommerce-search-qdrant.md b/qdrant-landing/content/blog/ecommerce-search-qdrant.md
new file mode 100644
index 000000000..d0443e134
--- /dev/null
+++ b/qdrant-landing/content/blog/ecommerce-search-qdrant.md
@@ -0,0 +1,97 @@
+---
+title: "Lessons From Building E-Commerce Search on Qdrant"
+draft: false
+slug: ecommerce-search-qdrant
+short_description: "We built e-commerce search over 5.8M real products on Qdrant. These are the ranking, personalization, and scaling decisions that generalize."
+description: "Build e-commerce search on Qdrant: lessons on hybrid ranking, in-query filtering, embedding recipes, personalization as re-rank, and merchandising formulas."
+preview_image: /blog/ecommerce-search-qdrant/hero.jpg
+social_preview_image: /blog/ecommerce-search-qdrant/hero.jpg
+date: 2026-07-24
+author: Dylan Couzon
+featured: true
+tags:
+ - ecommerce-search
+ - vector-search
+ - hybrid-search
+ - personalization
+ - recommendations
+---
+
+Relevance, filtering, personalization, merchandising, and recommendations usually arrive as five separate services, and the final ranking gets stitched across all of them. Each service ranks by its own rules, and none of them owns the order a shopper ends up seeing.
+
+We built [Qdrant Shopping](https://demo-ecommerce-search.vercel.app/), a storefront over 5.8 million real Amazon fashion products, to find out how many of those pieces collapse into one. Every text search is a single request to Qdrant's [Query API](/documentation/search/hybrid-queries/) that returns a ranked shelf in about 40 milliseconds, and the [code is on GitHub](https://github.com/qdrant-labs/demo-ecommerce-search). The decisions below apply to almost any product catalog, and we got several of them wrong before we got them right.
+
+
+
+## Start With Hybrid Retrieval
+
+A product query is two queries at once. Part of what a shopper types is an exact token: a brand, a size, a model number, a SKU, where the right result literally contains the string. The rest is intent, "something warm for hiking," where the right result may never use those words. Dense vectors match the intent and drift on the exact tokens. BM25 matches those tokens and has no way to reach "warm for hiking."
+
+So the storefront retrieves with both: a dense vector built from the product title and category, a BM25 sparse vector built from title, brand, and category, fused with reciprocal rank fusion in one Query API request. Both branches belong in the first version of a product search, before any relevance complaints arrive.
+
+## Filter Inside the Query
+
+Every category page, price bracket, size, and in-stock toggle is a filter, and the placement of that filter decides whether the page comes back full. Retrieve the top vector matches first and filter after, and a strict filter empties the page: you fetched 100 candidates, 90 were out of stock, and the shopper sees 10 results while better matches sit barely outside the window you pulled. Qdrant applies the filter during the search instead, walking its filterable HNSW graph so only matching products are ever scored, and the [facet counts](/documentation/search/filtering/) down the side of the page come from the same payload.
+
+Index every payload field you filter, sort, group, or reference in a formula. On Qdrant Cloud an unindexed field in any of those positions returns an error instead of a silent slow scan, so the schema has to declare the field before the first query uses it. Every team hits this once, and the error is cheaper than finding the missing index through production latency.
+
+## What You Embed Matters More Than Which Model
+
+A quality bug looked like a job for a bigger model: a search for a shirt was ranking on the brand rather than the garment. The fix turned out to be in the input text. The title alone gave the embedding too little to anchor on, so brand tokens dominated the vector. Adding the product category to the embedded text fixed the drift on every model we tried.
+
+We benchmarked the bigger model anyway. The table compares precision@10 on a 200,000-product subset with the recipe applied to both (scored by the LLM judge described below), then the latency and memory each model costs after re-ingesting the full 5.8 million products:
+
+| Model | Dimensions | precision@10 (subset) | Query latency | RAM (int8, full catalog) |
+|---|--:|--:|--:|--:|
+| MiniLM (shipped) | 384 | 90.3% | ~40 ms | 1.9 GB |
+| Larger model | 1024 | 92.3% | ~180 ms | 6 GB |
+
+Two points of precision@10 cost 4.5 times the query latency, 3 times the RAM, and an out-of-disk incident: the larger model's float32 originals filled a 96 GB node mid-ingest. We reverted. Embed the fields that define the product, its title, category, and key attributes, and check that text before you reach for a bigger model. The recipe is cheaper to change and usually fixes more.
+
+## Quantize in RAM, and Skip the Rescore
+
+The dense vectors live in RAM as int8, with the float32 originals on disk. Qdrant can [rescore](/documentation/manage-data/quantization/) the top candidates against those originals to undo quantization error, and we assumed a catalog this size would need it. Measured, the rescore bought 2 points of recall@10 at best while multiplying query time by 4.5 to 8, from 41 ms to as much as 334 ms, so we shipped without it.
+
+The rescore reads originals from disk, and reciprocal rank fusion damps the small ordering errors quantization introduces. Quantization noise matters when one score decides the final order. Here that score only feeds a rank fusion, where the small errors wash out, and the disk reads that correct them change nothing the shopper sees.
+
+## Personalize in Ranking, Not Retrieval
+
+Our first pass added the shopper's taste vector, built from their purchase history, as a third retrieval branch fused with the dense and keyword branches. A search for "jeans" could then return a flannel shirt, because the taste branch contributed candidates the query never asked for.
+
+So we moved taste out of retrieval and into ranking. The text query decides what qualifies; taste only reorders the page it returns, blended into the final rank at a weight of 0.45. Against the heavier 0.6 we started with, 0.45 held recall (0.82 versus 0.79) while cutting off-query items in the top 10 from 1.7 per query to 0.03. Below 0.45 the personas start converging on the same results with no further gain, so 0.45 is where both numbers hold and that is what we shipped.
+
+Two edge cases need a decision up front. A query with no text, a landing page, has no intent to protect, so taste drives the whole ranking there. A shopper with no history gets the plain query ranking, which is the right cold-start default.
+
+
+
+## Merchandising Is a Set of Weights
+
+Margin, popularity, freshness, price, rating, and stock are business signals, and the usual answer is a service that reorders search results according to them. We put them in the ranking instead. One [Formula Query](/documentation/search/hybrid-queries/) rescore runs after fusion and composes the final score from those payload fields, and a campaign like "clearance" or "new arrivals" is a set of weights on that formula rather than a separate code path.
+
+Presets and sliders change the weights, and the server clamps each to a fixed range. A merchandiser can retune a campaign without a deploy, and a bad slider value can't reshape the pipeline, only change how much a signal counts. That hands a merchandising desk to non-engineers without a second ranking service to keep in sync with search.
+
+
+
+## Every Eval Has a Blind Spot
+
+Our first relevance eval was a golden set: fixed expected product IDs per query. One recipe change took a 60-query fixture to 27% strict match overnight, and relevance had not dropped: the expected IDs were the old model's own output. The fixture measured distance from the model that built it, so every change to the pipeline scored as a regression.
+
+We replaced it with an LLM judge that asks, for each returned result, whether it is relevant to the query. It reads the results rather than an answer key, so it never anchors to any model, which is what made the model comparison above possible; the shipped pipeline scores 96% by this judge on the full catalog. But the judge is blind too: it only sees what came back, so it measures precision and says nothing about what retrieval missed.
+
+That is why the numbers in this post come from a small battery rather than one score. The judge scores precision, recall@10 catches what a cheaper setting drops, intrusion counts off-query items personalization sneaks into the top 10, and persona overlap checks that personalization still tells shoppers apart. Each one was added after the metric before it missed something. Even the golden set keeps a job, since it is cheap, deterministic, and reliable while the model stays frozen, and ours only broke once we started iterating. For any eval you run, name what it cannot see, then add the metric that covers it.
+
+## One Gotcha: Fusion and Shards
+
+If you shard the collection, watch for one Qdrant-specific trap. Fusion is global only when it is the main query. Nest it one level down inside a prefetch and each shard fuses its own local results, then the per-shard rankings merge. Wrapping a fused query in a rescore stage, exactly what the merchandising formula above does, is what pushes it down that level.
+
+Across four shards, a plain fused query reproduced the global top 10 on 87% of queries; the same query nested under a rescore fell to 55%. About half the top 10 comes back in a different position:
+
+
+
+If the catalog fits on one node, a single shard keeps fusion global and still lets you rescore in the same request. If it doesn't, keep the fusion as the main query so it stays global, retrieve a generous candidate set, and apply the merchandising formula as a separate pass over those results. Qdrant's [hybrid queries documentation](/documentation/search/hybrid-queries/) states the rule.
+
+## Try It
+
+The whole storefront (search, filters, personalization, merchandising, and product-page recommendations) runs on one Qdrant collection with product payloads and three vector types: a MiniLM dense vector, a BM25 sparse vector, and a CLIP image vector for visual similarity. Query embeddings run in-cluster through [Cloud Inference](/documentation/inference/cloud-inference/), so the app serves no embedding model of its own.
+
+[Qdrant Shopping is live](https://demo-ecommerce-search.vercel.app/), and the [full source](https://github.com/qdrant-labs/demo-ecommerce-search) shows the ingest, the schema, and the search path end to end. You can run the same pattern on a free [Qdrant Cloud](https://cloud.qdrant.io/signup) cluster and point it at your own catalog.
diff --git a/qdrant-landing/content/blog/facial-recognition.md b/qdrant-landing/content/blog/facial-recognition.md
index 356db7975..2236a9214 100644
--- a/qdrant-landing/content/blog/facial-recognition.md
+++ b/qdrant-landing/content/blog/facial-recognition.md
@@ -58,7 +58,7 @@ ___
## Application Workflows
-The app is divided into two phases - **The Offline Phase**, where the celebrity images are vectorized and **The Online Phase**, which carries out a live [**similarity search**]().
+The app is divided into two phases - **The Offline Phase**, where the celebrity images are vectorized and **The Online Phase**, which carries out a live **similarity search**.

diff --git a/qdrant-landing/content/blog/legal-tech-builders-guide.md b/qdrant-landing/content/blog/legal-tech-builders-guide.md
index 023d0bc0c..17f6639c4 100644
--- a/qdrant-landing/content/blog/legal-tech-builders-guide.md
+++ b/qdrant-landing/content/blog/legal-tech-builders-guide.md
@@ -95,13 +95,16 @@ Utilize [ColBERT](https://qdrant.tech/articles/late-interaction-models/) for hig
```python
# Step 1: Retrieve hybrid results using dense and sparse queries
-hybrid_results = client.search(
+hybrid_results = client.query_points(
collection_name="legal-hybrid-search",
- query_vector=dense_vector,
- query_sparse_vector=sparse_vector,
+ prefetch=[
+ models.Prefetch(query=dense_vector, using="dense", limit=50),
+ models.Prefetch(query=sparse_vector, using="sparse", limit=50),
+ ],
+ query=models.FusionQuery(fusion=models.Fusion.RRF),
limit=20,
with_payload=True
-)
+).points
# Step 2: Tokenize the query using ColBERT
colbert_query_tokens = colbert_model.query_tokenize(query_text)
@@ -111,7 +114,7 @@ reranked = sorted(
hybrid_results,
key=lambda doc: colbert_model.score(
colbert_query_tokens,
- colbert_model.doc_tokenize(doc["payload"]["document"])
+ colbert_model.doc_tokenize(doc.payload["document"])
),
reverse=True
)
@@ -195,4 +198,4 @@ The challenge isn’t just building something that works—it’s building somet
Successfully navigating LegalTech challenges requires careful balance across accuracy, compliance, scalability, and cost. Qdrant provides a comprehensive, flexible, and powerful vector search stack, empowering LegalTech to build robust and reliable AI applications.
-Ready to build? Start exploring Qdrant’s capabilities today through [Qdrant Cloud](https://cloud.qdrant.io/login) to strategically manage and advance your legal-tech applications.
\ No newline at end of file
+Ready to build? Start exploring Qdrant’s capabilities today through [Qdrant Cloud](https://cloud.qdrant.io/login) to strategically manage and advance your legal-tech applications.
diff --git a/qdrant-landing/content/blog/neural-search-tutorial.md b/qdrant-landing/content/blog/neural-search-tutorial.md
index f7093d052..a8cf91724 100644
--- a/qdrant-landing/content/blog/neural-search-tutorial.md
+++ b/qdrant-landing/content/blog/neural-search-tutorial.md
@@ -201,16 +201,16 @@ class NeuralSearcher:
vector = self.model.encode(text).tolist()
# Use `vector` for search for closest vectors in the collection
- search_result = self.qdrant_client.search(
+ search_result = self.qdrant_client.query_points(
collection_name=self.collection_name,
- query_vector=vector,
+ query=vector,
query_filter=None, # We don't want any filters for now
- top=5 # 5 the most closest results is enough
+ limit=5 # 5 the most closest results is enough
)
# `search_result` contains found vector ids with similarity scores along with the stored payload
# In this function we are interested in payload only
- payloads = [hit.payload for hit in search_result]
+ payloads = [hit.payload for hit in search_result.points]
return payloads
```
@@ -254,4 +254,4 @@ In this tutorial, I have tried to give minimal information about neural search,
Subscribe to my [telegram channel](https://t.me/neural_network_engineering), where I talk about neural networks engineering, publish other examples of neural networks and neural search applications.
-Subscribe to the [Qdrant user’s group](https://discord.gg/tdtYvXjC4h) if you want to be updated on latest Qdrant news and features.
\ No newline at end of file
+Subscribe to the [Qdrant user’s group](https://discord.gg/tdtYvXjC4h) if you want to be updated on latest Qdrant news and features.
diff --git a/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md b/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md
new file mode 100644
index 000000000..1f1e3aa68
--- /dev/null
+++ b/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md
@@ -0,0 +1,53 @@
+---
+title: "Pre-Filtering vs Post-Filtering (and Why Qdrant Does Neither)"
+draft: false
+slug: pre-filtering-vs-post-filtering
+short_description: "Pre-filtering degrades into brute force and post-filtering can return nothing. Qdrant filters during graph traversal and routes per query."
+description: "Compare pre-filtering vs post-filtering in vector search: where each breaks, how Qdrant filters in place, and when ACORN earns its cost."
+preview_image: /blog/pre-filtering-vs-post-filtering/preview/preview.jpg
+social_preview_image: /blog/pre-filtering-vs-post-filtering/preview/social_preview.jpg
+title_preview_image: /blog/pre-filtering-vs-post-filtering/preview/title.jpg
+date: 2026-08-07
+author: Dylan Couzon
+featured: false
+tags:
+ - vector-search
+ - filtering
+ - hnsw
+ - acorn
+---
+
+Adding a metadata filter to vector search can make good results disappear without making the query look broken. It still runs fast, returns something, and keeps the dashboards quiet, while some of the true nearest matches drop out. In the benchmark behind this post, a broad-value filter lowers recall to 90.8% and an `AND` filter over two broad values lowers it to 39.7%, while every other filter shape stays above 97%.
+
+The usual choice is between two strategies: pre-filtering, which applies the filter before the search, and post-filtering, which applies it after. That choice is simple at the extremes. The middle is the problem: a filter can match too many points for pre-filtering to stay cheap and too few for post-filtering to return anything.
+
+## Where Each Strategy Works
+
+Pre-filtering resolves the filter first: the engine computes the set of points that pass, usually as a mask over the whole collection, then searches within it. The filter is fully enforced, and scoring the matches directly makes the results exact for that subset. The cost grows fast: the mask touches every point, every match becomes a scoring candidate, and a broad filter degrades into brute force.
+
+Post-filtering searches first and filters the returned candidates after the fact. The engine compensates with an over-fetch, asking the nearest-neighbor search for more results than the query requested. That works for lenient filters, where most candidates pass, but strict filters can discard the whole set. The hard part is sizing it: undershoot and you return too little, overshoot and you drift back toward the brute-force work the index was meant to avoid.
+
+## What Qdrant Does Instead
+
+Qdrant runs the filter inside the search. A query walks the HNSW graph (Hierarchical Navigable Small World), the linked index that lets a search hop between neighbors instead of scanning the whole collection. Every candidate the traversal reaches is checked against the filter in place, and points that fail are skipped instead of scored.
+
+In-place filtering ties the cost to the traversal itself. It has one failure mode of its own: a strict filter leaves so few eligible points that the paths between them break, and the traversal dead-ends short of the best matches.
+
+Qdrant makes in-place filtering hold up with two repairs.
+
+- **[Filterable HNSW](/articles/filterable-hnsw/)** (2019): adds extra edges to the graph at index time between points that share a value in an indexed [payload field](/documentation/manage-data/indexing/#payload-index), so filtered queries keep connected paths to follow.
+- **ACORN** (2024): repairs the traversal at query time, reaching matches through neighbors that fail the filter.
+
+Extra edges cover most filter shapes, but they skip two cases by design. We have [a full article](/articles/filtered-vector-search-acorn/) on that gap: a value shared by too many points gets no extra edges, and a big tenant or a popular category is exactly that. An `AND` filter has no edges of its own even when each of its fields does, so an `AND` over two broad values falls into the same gap.
+
+The two failing shapes from the opening sit in that gap, and both recovered to 100% with ACORN on, measured in the article's default configuration.
+
+## How Qdrant Routes a Filtered Query
+
+Inside Qdrant, the [query planner](/documentation/search/search/#query-planning) settles the strategy question per query. It estimates how many points pass the filter and picks one of four paths: the filterable HNSW graph, the same graph with ACORN, the payload index directly, or a full scan. The payload-index path stays cheap because the index already lists which points match. On one `AND` filter from the article, matching 1% of points, 471 of 500 queries read the payload index while 29 walked the graph.
+
+{{< figure src="/blog/pre-filtering-vs-post-filtering/query-routing.svg" alt="A filtered query flows into the query planner, which estimates how many points pass the filter and routes to the HNSW graph, holding extra edges and opt-in ACORN traversal, to the payload index, or to a full scan. The arrow to the payload index is tagged 471 of 500 and the arrow to the graph 29 of 500." caption="The four paths the planner picks from. The tagged counts are that 1% `AND` filter's routing split." width="100%" >}}
+
+For a deeper pass on testing your own collection, see [the full article](/articles/filtered-vector-search-acorn/). It covers `exact: true` for brute-force ground truth and when [`acorn.enable`](/documentation/search/search/#acorn-search-algorithm) is worth turning on: a few times the latency when a query runs on the graph, almost nothing when the planner routes it to the payload index.
+
+Engines split on this choice: [some post-filter, others pre-filter](/benchmarks/filtered-search-benchmark/), and the choice becomes part of their architecture. Qdrant made it a planning decision instead, settled per query. Filtering belongs inside the search, where the engine has enough context to pick the path.
diff --git a/qdrant-landing/content/blog/qdrant-1.16.x.md b/qdrant-landing/content/blog/qdrant-1.16.x.md
index e6b8d0492..27a6df012 100644
--- a/qdrant-landing/content/blog/qdrant-1.16.x.md
+++ b/qdrant-landing/content/blog/qdrant-1.16.x.md
@@ -32,7 +32,7 @@ Additionally, version 1.16 introduces a new conditional update API, facilitating
Multitenancy is a common requirement for SaaS applications, where multiple customers (tenants) share the same database instance. In Qdrant, when an instance is shared between multiple users, you may need to partition vectors by user. This is done so that each user can only access their own vectors and can’t see the vectors of other users. To implement multitenancy in Qdrant, there are two main approaches:
- [Payload-based multitenancy](/documentation/manage-data/multitenancy/), which works well when you have a large number of small tenants. This causes practically no overhead. Quite the opposite: a query with a tenant payload filter can be faster than a full search.
-- [Shard-based multitenancy](/documentation/distributed_deployment/#user-defined-sharding), designed for when you have a smaller number of larger tenants. This works well when each tenant requires isolation and dedicated resources. Separating tenants by shard prevents a classic noisy neighbor problem where a single high-volume tenant can force the cluster to scale for everyone, increasing costs and reducing performance for smaller tenants. However, shard-based multitenancy is not a good solution when you have a large number of small tenants, as each shard incurs some overhead.
+- [Shard-based multitenancy](/documentation/scaling/distributed_deployment/#user-defined-sharding), designed for when you have a smaller number of larger tenants. This works well when each tenant requires isolation and dedicated resources. Separating tenants by shard prevents a classic noisy neighbor problem where a single high-volume tenant can force the cluster to scale for everyone, increasing costs and reducing performance for smaller tenants. However, shard-based multitenancy is not a good solution when you have a large number of small tenants, as each shard incurs some overhead.
Real-world usage patterns often fall between these two use cases. It's common to have a small number of large tenants and a huge tail of smaller ones. You may even have tenants that grow over time, starting small and eventually becoming large enough to require dedicated resources.
@@ -40,7 +40,7 @@ In version 1.16, Qdrant can now efficiently combine the two multitenancy approac
The main principles behind Tiered Multitenancy are:
-- [User-defined Sharding](/documentation/distributed_deployment/#user-defined-sharding) allows you to create named shards within a collection. It enables you to isolate large tenants into their own shards. A multitenant collection can consist of a shared "fallback" shard for small tenants and multiple dedicated shards for large tenants.
+- [User-defined Sharding](/documentation/scaling/distributed_deployment/#user-defined-sharding) allows you to create named shards within a collection. It enables you to isolate large tenants into their own shards. A multitenant collection can consist of a shared "fallback" shard for small tenants and multiple dedicated shards for large tenants.
- **Fallback shards** - a special routing mechanism that allows Qdrant to route a request to either a dedicated shard (if it exists) or to a shared fallback shard. This keeps requests unified, without the need to know whether a tenant is dedicated or shared.
- [Tenant promotion](/documentation/manage-data/multitenancy/#promote-tenant-to-dedicated-shard) - a mechanism that makes it possible to "promote" tenants from the shared Fallback Shard to their own dedicated shard when they grow large enough. This process is based on Qdrant’s internal shard transfer mechanism, which makes promotion completely transparent for the application. Both read and write requests are supported during the promotion process.
diff --git a/qdrant-landing/content/blog/qdrant-1.17.x.md b/qdrant-landing/content/blog/qdrant-1.17.x.md
index dd3681c99..da81a25b7 100644
--- a/qdrant-landing/content/blog/qdrant-1.17.x.md
+++ b/qdrant-landing/content/blog/qdrant-1.17.x.md
@@ -115,7 +115,7 @@ Many people have been asking about point filtering in web UI. And now it's back,
As an open source project, we welcome contributions from the Qdrant community. This release features two contributions from community members:
- Not all payload field indexes are used in combination with dense vector queries. With this release, you can [specify whether individual payload field indexes should be reflected in the HNSW index](/documentation/manage-data/indexing/#disable-the-creation-of-extra-edges-for-payload-fields).
-- A new API endpoint is available to [list all user-defined shard keys](/documentation/distributed_deployment/#user-defined-sharding).
+- A new API endpoint is available to [list all user-defined shard keys](/documentation/scaling/distributed_deployment/#user-defined-sharding).
Additionally, this release adds the following features:
diff --git a/qdrant-landing/content/blog/qdrant-1.19.x.md b/qdrant-landing/content/blog/qdrant-1.19.x.md
new file mode 100644
index 000000000..f0940bc8b
--- /dev/null
+++ b/qdrant-landing/content/blog/qdrant-1.19.x.md
@@ -0,0 +1,141 @@
+---
+title: "Qdrant 1.19 - TurboQuant Datatype & Memory Tiers"
+draft: false
+slug: qdrant-1.19.x
+short_description: "Version 1.19 of Qdrant introduces the TurboQuant datatype, a new storage format that reduces disk usage by up to nine times."
+description: "Version 1.19 of Qdrant introduces the TurboQuant datatype for major storage savings, unified memory tier configuration, replica read affinity, and full-text search enhancements."
+preview_image: /blog/qdrant-1.19.x/social_preview.jpg
+social_preview_image: /blog/qdrant-1.19.x/social_preview.jpg
+date: 2026-08-05T00:00:00-01:00
+author: Andrey Vasnetsov
+featured: true
+tags:
+ - vector search
+ - quantization
+ - memory management
+ - full-text search
+---
+
+[**Qdrant 1.19.0 is out!**](https://github.com/qdrant/qdrant/releases/tag/v1.19.0) Let's look at the main features for this version:
+
+**TurboQuant Datatype:** A new storage format that compresses vectors to four bits without keeping their original full-precision representation, reducing storage by up to nine times compared to TurboQuant quantization.
+
+**Memory Tiers:** A single `memory` parameter unifies per-component memory tier placement, with three tiers: `pinned`, `cached`, and `cold`.
+
+**Per-Tenant IDF Statistics:** Narrow the IDF corpus to a specific tenant so term rarity reflects that tenant's vocabulary rather than the whole dataset, improving BM25 scoring in multi-tenant deployments.
+
+**Filtering Enhancements:** Prefix matching on keyword fields and a new slice filter condition for partitioning a collection's points into deterministic, disjoint subsets.
+
+**Web UI Enhancements:** Live resharding progress, an overhauled Collection Visualizer that scales to tens of thousands of points, and payload index management.
+
+## TurboQuant Datatype
+
+
+
+Version 1.18 introduced [TurboQuant](/documentation/manage-data/quantization/#turboquant-quantization), a quantization method that compresses vectors with minimal loss in recall. It operates as a secondary layer: Qdrant keeps the original full-precision vectors on disk alongside the compressed copy, using the quantized representation during HNSW traversal and rescoring against the original vectors for accuracy. The two-copy model delivers good recall, but storing both representations increases disk usage.
+
+In version 1.19, we've applied the same method to storage itself. The new [Turbo4 datatype](/documentation/manage-data/vectors/#turbo4) stores vectors using 4-bit TurboQuant compression as the only representation, with no full-precision copy kept.
+
+Storing only the 4-bit representation drops storage from 36 bits per coordinate (the full float32 original plus the 4-bit compressed copy) to four bits, resulting in a ninefold reduction. This reduces data reads and writes per operation, improving throughput. The same compression applies to multi-vector collections used for ColBERT-style late interaction search, where the benefit is proportionally larger.
+
+That ninefold storage reduction comes at a cost: without a full-precision copy, Qdrant cannot rescore top candidates against the original vectors. This makes the TurboQuant datatype the right choice when reducing disk usage is the primary goal. When maximum recall is the priority, TurboQuant quantization on top of a full-precision storage type remains the better option.
+
+## Memory Tiers
+
+
+
+A Qdrant collection stores data across several components, each with its own memory footprint: vectors, the HNSW index, quantized vectors, the sparse index, payloads, and payload indexes. Until now, each had its own way to configure whether that data is loaded into RAM or served from disk: `on_disk`, `always_ram`, and `on_disk_payload`. This release replaces these parameters with [a single, unified `memory` parameter](/documentation/ops-configuration/memory-tiers/). It works the same way on every component, giving you one consistent way to configure the memory tier for any part of a collection.
+
+There are three memory tiers: `pinned` loads the component entirely into memory, where it's never evicted (for components that support it); `cached` keeps data on disk and pre-populates the OS disk cache at startup for fast first reads while remaining evictable under memory pressure; and `cold` loads it lazily from disk on first access. The existing per-component flags remain functional but are deprecated.
+
+Beyond cleaner configuration, version 1.19 also adds new capabilities: HNSW graph links can now be pinned in memory, sparse indexes have gained a new `cached` tier, and quantized vectors can now be pinned, cached, or cold independently of the original vectors' placement.
+
+## Per-Tenant IDF Statistics
+
+
+
+Sparse vector search commonly uses the inverse document frequency (IDF) to score matching documents, giving rarer terms more weight than common ones. Calculating the IDF requires two statistics: the total number of documents and the number of documents containing each term.
+
+Qdrant computes these statistics for the complete dataset in each shard being queried, which creates a problem for multi-tenant collections. If tenant A's documents use different vocabulary than tenant B's, blending both populations into one set of statistics distorts a term's IDF, so it no longer reflects how rare that term is within either tenant's data.
+
+Version 1.19 lets you [narrow the corpus that the IDF statistics are computed over](/documentation/manage-data/multitenancy/#per-tenant-idf-statistics), for example down to a single tenant, so the IDF reflects term rarity within that tenant's data rather than the whole collection.
+
+## Filtering Enhancements
+
+
+
+This release adds two new filtering capabilities to Qdrant: prefix matching on keyword fields, and a slice filter condition for partitioning a collection's points into deterministic subsets.
+
+### Prefix Matching on Keyword Fields
+
+Keyword indexes store values verbatim for exact matching, which is the right choice for identifiers like URLs, file paths, and SKUs. Filtering by prefix over these values, like *"find all entries where the URL starts with `https://qdrant.`"*, wasn't possible without either a full payload scan or switching to a text index, which tokenizes values and breaks exact matching.
+
+This release adds [support for prefix matching to keyword indexes](/documentation/manage-data/indexing/?q=indexing#prefix-matching-in-keyword-indices). Enable it with `"prefix": true` in the keyword index configuration, then use [the `prefix` condition](/documentation/search/filtering/#prefix-match) in your filter. Prefix queries are served from a dedicated index structure, making them as fast as any other indexed filter.
+
+### Slicing
+
+The new [slice filter condition](/documentation/search/filtering/#slice) groups a collection's points into deterministic, disjoint subsets. Each slice selects a fixed, stable portion of the collection without overlap.
+
+This opens up two patterns that were previously difficult to implement efficiently. For parallel processing, divide a collection into `n` slices and assign one worker per slice. This lets you scroll the full dataset concurrently without coordination between workers. For reproducible sampling, use the same slice across multiple queries. The same subset of points is always selected, making it straightforward to benchmark, test, or run experiments on a consistent portion of your data.
+
+## Web UI Enhancements
+
+
+
+The [Web UI](/documentation/web-ui/) is Qdrant's user interface for managing deployments and collections. It enables you to create and manage collections, run API calls, import sample datasets, and learn about Qdrant's API through interactive tutorials. In version 1.19, the Web UI has gained several new features.
+
+### Resharding Progress
+
+[Resharding](/documentation/scaling/distributed_deployment/#resharding) changes the number of shards for a collection, a process that can take a long time on large collections. Previously, the Web UI only showed that resharding was running, without visibility into its progress.
+
+The Collection **Cluster** tab now displays a live progress message for the duration of the operation. It names the shards being added or removed, and shows the current stage.
+
+
+
+### Collection Visualizations
+
+The Collection **Visualize** tab shows interactive 2D visualizations of your vectors, so you can visually explore your data and see how it clusters.
+
+In this release, the Collection Visualizer has moved from a browser-side pipeline to a server-driven one. Qdrant now computes distances server-side, instead of the browser downloading raw vectors, and the layout engine runs in WebAssembly for a much more responsive experience. Together, these changes raise the practical point limit from a few thousand to tens of thousands, and a new WebGL2 renderer keeps panning and zooming smoothly at that scale.
+
+The Collection Visualizer also gains new ways to explore a collection: click a point to see its nearest neighbors highlighted, or Shift+drag to select a region of points to open a selection panel listing them, with one-click copy for their IDs, JSON, or a matching filter. You can also apply a filter to highlight matching points.
+
+
+
+### Payload Index Configuration
+
+Previously, you could only create and manage [payload indexes](/documentation/manage-data/indexing/#payload-index) through Qdrant's API. The Web UI now lets you do this interactively: hover over any payload field in the Collections **Points** tab and click the index icon to configure an index on that field. Qdrant suggests the index type automatically from the field value, and type-specific options appear where applicable: the tokenizer and phrase matching settings for `text` indexes, or the range and lookup toggles for `integer` indexes.
+
+The Collection **Info** tab now includes a payload indexes overview that lists all indexed fields with their types, and lets you edit or delete any of them from one place.
+
+
+
+## Also in This Release
+
+
+
+- **[Resource Quotas](/documentation/ops-configuration/quotas/)**: Prevent nodes from running out of memory or disk space by limiting how much memory and disk they can use. This supersedes the per-collection `max_resident_memory_percent` strict mode setting, which is now deprecated.
+- **[Replica Read Affinity](/documentation/scaling/consistency-guarantees/#read-affinity)**: Provide an `X-Qdrant-Route-Affinity` HTTP header with a user or session ID to pin that user's reads to the same replica, eliminating read inconsistency when sequential reads land on different replicas.
+- **[BM25: Language-Neutral Text Processing](/documentation/search/text-search/full-text-search/#language-neutral-text-processing)**: Turn off English stemming and stopword removal in BM25 text processing for a clean language-neutral text search pipeline, better suited to technical content, product identifiers, or multilingual text.
+- **Faster Faceting**: [Faceting](/documentation/manage-data/payload/#facet-counts) is a query feature that counts how many points match each distinct value of a payload field within a filtered result set. In 1.19, facet queries are faster, especially on large collections with high-cardinality fields.
+- **Removal of deprecated endpoints**: We've removed the legacy `/search`, `/recommend`, and `/discover` endpoints. The unified `/query` API [superseded these endpoints in version 1.10](/blog/qdrant-1.10.x/#one-endpoint-for-all-queries). If you still use these endpoints, migrate to the [`/query` API](/documentation/search/search/#query-api) before upgrading to 1.19.
+
+For a full list of all changes in version 1.19, see the [changelog](https://github.com/qdrant/qdrant/releases/tag/v1.19.0).
+
+## Upgrading to Version 1.19
+
+
+
+On Qdrant Cloud, navigate to the Cluster Details screen and select Version 1.19 from the dropdown menu. The upgrade process may take a few moments.
+
+We recommend upgrading versions one by one. Qdrant Cloud does this automatically when you select the target version. If you're self-hosting, upgrade to the latest patch version of each intermediate minor version first, for example, 1.17.x→1.18.x→1.19.0.
+
+> If you still use the legacy `/search`, `/recommend`, or `/discover` endpoints, migrate to the [`/query` API](/documentation/search/search/#query-api) before upgrading to 1.19.
+
+Need help with your upgrade? The [Qdrant Advisor agent skill](https://qdrant.tech/documentation/skills/) can help you navigate upgrades, troubleshoot configurations, and answer questions about your Qdrant setup, whether you're on Qdrant Cloud or self-hosting.
+
+## Engage
+
+
+
+We would love to hear your thoughts on this release. If you have any questions or feedback, join our [Discord](https://discord.gg/qdrant) or create an issue on [GitHub](https://github.com/qdrant/qdrant/issues).
diff --git a/qdrant-landing/content/blog/qdrant-1.9.x.md b/qdrant-landing/content/blog/qdrant-1.9.x.md
index dc3eee261..f423554c4 100644
--- a/qdrant-landing/content/blog/qdrant-1.9.x.md
+++ b/qdrant-landing/content/blog/qdrant-1.9.x.md
@@ -40,7 +40,7 @@ We highly recommend this feature to enterprises using [Qdrant Hybrid Cloud](/hyb
## Faster shard transfers on node recovery
-We now offer a streamlined approach to [data synchronization between shards](/documentation/distributed_deployment/#shard-transfer-method) during node upgrades or recovery processes. Traditional methods used to transfer the entire dataset, but our new `wal_delta` method focuses solely on transmitting the difference between two existing shards. By leveraging the Write-Ahead Log (WAL) of both shards, this method selectively transmits missed operations to the target shard, ensuring data consistency.
+We now offer a streamlined approach to [data synchronization between shards](/documentation/scaling/distributed_deployment/#shard-transfer-method) during node upgrades or recovery processes. Traditional methods used to transfer the entire dataset, but our new `wal_delta` method focuses solely on transmitting the difference between two existing shards. By leveraging the Write-Ahead Log (WAL) of both shards, this method selectively transmits missed operations to the target shard, ensuring data consistency.
In some cases, where transfers can take hours, this update **reduces transfers down to a few minutes.**
@@ -48,7 +48,7 @@ The advantages of this approach are twofold:
1. **It is faster** since only the differential data is transmitted, avoiding the transfer of redundant information.
2. It upholds robust **ordering guarantees**, crucial for applications reliant on strict sequencing.
-For more details on how this works, check out the [shard transfer documentation](/documentation/distributed_deployment/#shard-transfer-method).
+For more details on how this works, check out the [shard transfer documentation](/documentation/scaling/distributed_deployment/#shard-transfer-method).
> **Note:** There are limitations to consider. First, this method only works with existing shards. Second, while the WALs typically retain recent operations, their capacity is finite, potentially impeding the transfer process if exceeded. Nevertheless, for scenarios like rapid node restarts or upgrades, where the WAL content remains manageable, WAL delta transfer is an efficient solution.
diff --git a/qdrant-landing/content/blog/qdrant-academy-launch.md b/qdrant-landing/content/blog/qdrant-academy-launch.md
index 17509756e..4d79ff740 100644
--- a/qdrant-landing/content/blog/qdrant-academy-launch.md
+++ b/qdrant-landing/content/blog/qdrant-academy-launch.md
@@ -69,7 +69,6 @@ Our current partner content tutorials include:
* [Tensorlake](https://qdrant.tech/course/essentials/day-7/tensorlake/)
* [LlamaIndex](https://qdrant.tech/course/essentials/day-7/llamaindex/)
* [Unstructured.io](https://qdrant.tech/course/essentials/day-7/unstructured/)
-* [Quotient](https://qdrant.tech/course/essentials/day-7/quotient/)
* [Superlinked](https://qdrant.tech/course/essentials/day-7/superlinked/)
* [Camel AI](https://qdrant.tech/course/essentials/day-7/camel/)
* [Jina AI](https://qdrant.tech/course/essentials/day-7/jina/)
diff --git a/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md b/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md
new file mode 100644
index 000000000..f0de9258e
--- /dev/null
+++ b/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md
@@ -0,0 +1,84 @@
+---
+title: "Enough with the Bad Benchmarks: Tools for Production-Grade Research"
+draft: false # TODO: flip to false when ready to publish
+slug: qdrant-fineweb-10b-release
+short_description: "Qdrant releases Qdrant-FineWeb-10B, a 10-billion vector benchmark dataset, alongside Supernova, an open-source internet-scale benchmarking engine."
+description: "Qdrant and Vultr release Qdrant-FineWeb-10B, the largest open-source vector search benchmark, plus Supernova, an open-source framework for embedding generation, ground truth, and evaluation at billion-vector scale."
+preview_image: /blog/qdrant-fineweb-10b-release/Blog-Hero.png # TODO: add preview image to static/blog/qdrant-fineweb-10b-release/
+social_preview_image: /blog/qdrant-fineweb-10b-release/Blog-Hero.png # TODO: add social preview image
+date: 2026-09-01
+author: Qdrant Labs
+featured: true
+tags:
+ - Benchmarks
+ - Datasets
+ - Open Source
+---
+
+Real world vector search workloads are increasingly large and complex. Enterprises are not using vector search to occasionally search through a couple of PDF files. They are indexing and searching billions of vectors at thousands of requests per second (RPS) and sub 50ms tail latency. Large enterprises also can’t tolerate faulty assumptions.
+
+Too many benchmarks use gated, proprietary managed services. And even worse, the data is synthetic, the queries are hidden, and the engines are locked behind paywalls.
+
+These benchmarks show a 90% recall @ 10 on a purely synthetic RAG benchmark for a production claim. But for an e-commerce or marketplace company feeding top 1000+ results into a second-stage reranker, it’s not good enough. Engineers within the world’s leading search teams demand reproducibility, 95%+ recall, large retrieval depth, high throughput, and sub 100-ms p99 latency. And so do we.
+
+Why hasn’t this problem already been solved? Because [generating benchmark datasets at billion-scale is incredibly challenging](https://openreview.net/forum?id=8MhuCdCECA), and calculating exact ground truth queries requires vast compute and potentially *quadrillions* of brute-force distance computations.
+
+But we like hard problems. So we went after it.
+
+In partnership with [Vultr](https://www.vultr.com/), we absorbed the economics of extreme-scale embedding generation and brute-force KNN to provide the scientific and engineering communities with a new standard of search benchmarking datasets: [**Qdrant-FineWeb-10B**](https://huggingface.co/datasets/Qdrant/FineWeb-10B).
+
+This behemoth of a dataset contains \~25 TiB of vector data alone, consisting of **dense and sparse vectors** generated from [`gte-multilingual-base`](https://huggingface.co/Alibaba-NLP/gte-multilingual-base). Then, using our GPU-native benchmarking engine, we computed the exact top-1000 ground truth for 120,000 dense, sparse, and filtered queries \- over a quadrillion distance computations across the full 10B document corpus.
+
+Because the industry currently lacks datasets that translate well to multimodal formats and complex filtering, we are also releasing [**PubMed-Multi-Vector**](https://huggingface.co/datasets/Qdrant/PubMed-MV) and [**Coyo-Vector-Embeddings**](https://huggingface.co/datasets/Qdrant/Coyo-VE). These datasets provide the community with the dense, sparse, and multimodal representations that actually reflect modern production architectures.
+
+To ensure these datasets aren't just another proprietary vendor claim, we are open-sourcing the tooling that we used. [**Supernova**](https://github.com/qdrant-labs/supernova) is our high-performance, distributed benchmarking framework designed to make massive-scale dataset generation and brute-force ground-truthing accessible, reproducible and even more cost-effective.
+
+The era of relying on 1-million vector datasets, production hearsay, and synthetic approximations is over. Here is the real data, the real ground truth, and the open-source infrastructure you need to run it yourself.
+
+## Qdrant-FineWeb-10B: Benchmarking at Internet Scale
+
+**Qdrant-FineWeb-10B** represents the core of this release. It is the largest open-source vector search benchmark available to the community, comprising **24.47 TB of vectors** and **28.66 TB of source text and metadata**.
+
+Created in collaboration with Vultr using `gte-multilingual-base` on Hugging Face's FineWeb corpus, we utilized Supernova to compute exact top-1000 brute-force ground-truth nearest neighbors for 100,000 queries across the entire 10-billion vector space. Taken together, this amounted to over **one quadrillion distance calculations** run in parallel on GPU-accelerated hardware.
+
+### Additional Community Datasets
+
+To showcase Supernova’s versatility across modalities and provide further assets to the community, we used Supernova to generate two additional open datasets:
+
+| Dataset | Model | Data Type | Vectors | Ground Truth |
+| :---- | :---- | :---- | :---- | :---- |
+| [**Qdrant-FineWeb-10B**](http://huggingface.co/datasets/Qdrant/FineWeb-10B) | `gte-multilingual-base` | Text | 10.07B dense, 10.07B sparse | Dense, sparse, filtered |
+| [**PubMed-Multi-Vector**](https://huggingface.co/datasets/Qdrant/PubMed-MV) | `BGE-M3` | Text | 23.9M dense, 23.9M sparse, 8.37B multi-vector tokens | Dense, sparse, multi-vector |
+| [**Coyo-Vector-Embeddings**](https://huggingface.co/datasets/Qdrant/Coyo-VE) | `Qwen3-VL-Embedding-2B` | Text & Images | 15.4M dense (2048-dim) | Dense |
+
+* **PubMed-Multi-Vector**: Designed to benchmark hybrid retrieval methods with corpus variables held constant. It generates dense, sparse, and ColBERT-style multi-vector representations over the exact same text corpus, accumulating over 8.37 billion multi-vector tokens across nearly 35 TB of data.
+* **Coyo-Vector-Embedding**: Focuses on multimodal retrieval, leveraging a 2048-dimensional vision-language encoder (`Qwen3-VL-Embedding-2B`) to project image-caption pairs from the LLaVA dataset into a unified shared embedding space. This represents a highly-modern workload that leverages state of the art embedding generation and model architectures.
+
+We plan to continue to release more datasets for the community.
+
+---
+
+## Supernova: The Open-Source Benchmarking Engine
+
+We didn't want to stop at releasing a static dataset. We built **Supernova** as a free, fully open-source framework so that the broader community can generate, manipulate, ground-truth, and benchmark internet-scale datasets on their own infrastructure, without relying on Qdrant or any third-party stack.
+
+[Supernova](https://github.com/qdrant-labs/supernova) automates the four core phases of building and running a vector search benchmark: **1\) embedding generation, 2\) brute-force ground-truth calculation, 3\) database loading, and 4\) evaluation benchmarking.**
+
+Each phase is driven by a specialized module configured entirely via YAML files and designed for massively parallel execution:
+
+* **`nova-embed` (Modular Embedding Pipeline)**: Unifies disparate backends (SentenceTransformers, FastEmbed, OpenAI APIs) and storage systems (Hugging Face, S3, Cloudflare R2). It operates statelessly without a central database—each worker uses its rank and world size to partition input data independently, achieving linear scaling across cloud and HPC environments.
+* **`nova-bf` (GPU-Native Ground Truth)**: Computes exact brute-force top-$k$ nearest neighbors across dense, sparse, and multi-vector representations without running out of memory. It streams data partitions from remote storage, uses custom fused GPU kernels for late-interaction scoring, and evaluates filters early on the CPU to prune irrelevant rows before GPU transfer.
+* **`nova-load` & `nova-storm` (Ingestion & Stress Testing)**: Handle downstream evaluation across backends such as Qdrant, Milvus, and Elasticsearch. `nova-load` drives parallel ingestion to test write throughput, while `nova-storm` runs search workloads to track QPS, latency distributions ($p\_{50}, p\_{95}, p\_{99}$), build times, and recall accuracy against `nova-bf` ground truth.
+
+
+Overview of the Supernova framework: a standardized pipeline for vector search benchmarking. Each module is designed to scale linearly across cloud and HPC environments, enabling reproducible benchmarking at internet scale.
+
+### Distributed Compute with SkyPilot
+
+To scale compute seamlessly across distributed infrastructure, Supernova integrates `nova-dist`, a controller-only module built on [**SkyPilot**](https://skypilot.ai/) that handles cluster provisioning, job scheduling, fault tolerance, and cloud abstraction. Rather than hardcoding infrastructure logic into individual pipeline modules, `nova-dist` decouples job execution from hardware management. This allows `nova-embed`, `nova-bf`, `nova-load`, and `nova-storm` to scale linearly across AWS, GCP, Azure, Kubernetes, and Slurm HPC clusters using identical YAML configurations—massively parallelizing workloads across hundreds of GPUs without manual infrastructure overhead.
+
+---
+
+## Acknowledgements
+
+We extend our sincere thanks to **Vultr** for providing the raw compute infrastructure to generate the initial FineWeb-10B embeddings, as well as the **SkyPilot** and **Hugging Face** teams for building open-source foundation tools that enable operating at this scale.
\ No newline at end of file
diff --git a/qdrant-landing/content/blog/qdrant-relari.md b/qdrant-landing/content/blog/qdrant-relari.md
index 4638e1c56..d01cff6bb 100644
--- a/qdrant-landing/content/blog/qdrant-relari.md
+++ b/qdrant-landing/content/blog/qdrant-relari.md
@@ -21,7 +21,7 @@ tags:
Evaluating the performance of a [Retrieval-Augmented Generation (RAG)](/rag/) application can be a complex task for developers.
-To help simplify this, Qdrant has partnered with [Relari](https://www.relari.ai) to provide an in-depth [RAG evaluation](/articles/rapid-rag-optimization-with-qdrant-and-quotient/) process.
+To help simplify this, Qdrant has partnered with [Relari](https://www.relari.ai) to provide an in-depth RAG evaluation process.
As a [vector database](https://qdrant.tech), Qdrant handles the data storage and retrieval, while Relari enables you to run experiments to assess how well your RAG app performs in real-world scenarios. Together, they allow for fast, iterative testing and evaluation, making it easier to keep up with your app's development pace.
diff --git a/qdrant-landing/content/blog/rag-evaluation-guide.md b/qdrant-landing/content/blog/rag-evaluation-guide.md
index abc574498..28557e372 100644
--- a/qdrant-landing/content/blog/rag-evaluation-guide.md
+++ b/qdrant-landing/content/blog/rag-evaluation-guide.md
@@ -53,7 +53,7 @@ Additionally, the [“Lost in the Middle”](https://arxiv.org/abs/2307.03172) p

-To simplify the evaluation process, several powerful frameworks are available. Below we will explore three popular ones: **Ragas, Quotient AI, and Arize Phoenix**.
+To simplify the evaluation process, several powerful frameworks are available. Below we will explore two popular ones: **Ragas and Arize Phoenix**.
### Ragas: Testing RAG with questions and answers
@@ -63,19 +63,11 @@ To simplify the evaluation process, several powerful frameworks are available. B

-### Quotient: evaluating RAG pipelines with custom datasets
-
-Quotient AI is another platform designed to streamline the evaluation of RAG systems. Developers can upload evaluation datasets as benchmarks to test different prompts and LLMs. These tests run as asynchronous jobs: Quotient AI automatically runs the RAG pipeline, generates responses and provides detailed metrics on faithfulness, relevance, and semantic similarity. The platform's full capabilities are accessible via a Python SDK, enabling you to access, analyze, and visualize your Quotient evaluation results to discover areas for improvement.
-
-**Figure 2:** *Output of the Quotient framework, with statistics that define whether the dataset is properly manipulated throughout all stages of the RAG pipeline: indexing, chunking, search and context relevance.*
-
-
-
### Arize Phoenix: Visually Deconstructing Response Generation
[Arize Phoenix](https://docs.arize.com/phoenix) is an open-source tool that helps improve the performance of RAG systems by tracking how a response is built step-by-step. You can see these steps visually in Phoenix, which helps identify slowdowns and errors. You can define "[evaluators](https://arize.com/docs/phoenix/evaluation/concepts-evals/evaluators)" that use LLMs to assess the quality of outputs, detect hallucinations, and check answer accuracy. Phoenix also calculates key metrics like latency, token usage, and errors, giving you an idea of how efficiently your RAG system is working.
-**Figure 3:** *The Arize Phoenix tool is intuitive to use and shows the entire process architecture as well as the steps that take place inside of retrieval, context and generation.*
+**Figure 2:** *The Arize Phoenix tool is intuitive to use and shows the entire process architecture as well as the steps that take place inside of retrieval, context and generation.*

@@ -97,7 +89,7 @@ Vector databases support different [indexing](https://qdrant.tech/documentation/
**Develop a proper chunking/text splitting strategy**: Make sure your chunking/text splitting strategy is tailored to your on data type (e.g., HTML, markdown, code, PDF) and use-case nuances. For example, legal documents may be split by headings and subsections, and medical literature by sentence boundaries or key concepts.
-**Figure 4:** *You can use utilities like [ChunkViz](https://chunkviz.up.railway.app/) to visualize different chunk splitting strategies, chunk sizes, and chunk overlaps.*
+**Figure 3:** *You can use utilities like [ChunkViz](https://chunkviz.up.railway.app/) to visualize different chunk splitting strategies, chunk sizes, and chunk overlaps.*

@@ -180,7 +172,7 @@ First, create question and ground-truth answer pairs from source documents for t
Once you have created a dataset, collect the retrieved context and the final answer generated by your RAG pipeline for each question.
-**Figure 5:** *Here is an example of four evaluation metrics:*
+**Figure 4:** *Here is an example of four evaluation metrics:*
- **question**: A set of questions based on the source document.
- **ground_truth**: The anticipated accurate answers to the queries.
diff --git a/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md b/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md
new file mode 100644
index 000000000..9b072fba2
--- /dev/null
+++ b/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md
@@ -0,0 +1,73 @@
+---
+title: "How to Tune Vector Search Without Guessing"
+draft: false
+slug: tuning-retrieval-which-knob-first
+short_description: "The reference tells you what each retrieval setting does. Five measurements tell you what your own data needs, and we ran all five."
+description: "Tune retrieval in Qdrant: the measurement behind fusion k, candidate depth, rerankers, rescoring, and labeled set size."
+preview_image: /blog/tuning-retrieval-which-knob-first/hero.jpg
+social_preview_image: /blog/tuning-retrieval-which-knob-first/hero.jpg
+date: 2026-08-24
+author: Dylan Couzon
+featured: true
+weight: 0 # Change this weight to change order of posts
+tags:
+ - retrieval tuning
+ - hybrid search
+ - search relevance
+ - reranking
+ - quantization
+---
+
+Your collection works. Queries return in a few milliseconds, results are mostly right, and product keeps forwarding you the ones that aren't. You open the search API reference and get exact definitions for `hnsw_ef`, reciprocal rank fusion `k`, and quantization `oversampling`. The definitions are correct. They still don't tell you which setting is failing on your data.
+
+So you change one setting, rerun the queries, and the score moves by 0.01. Did relevance improve, or did the same queries land differently?
+
+Each setting has a right value, and it depends on something about your collection that no default can see. That something is measurable. We ran those measurements on five public datasets, from 5,183 to 4.6 million documents, and published the results today in five articles. Here's the problem each one solves, and the result we didn't expect.
+
+The five articles work on one query path. A dense and a sparse prefetch retrieve candidates, fusion merges the two lists into one ranking, and an optional reranker reorders the top of it.
+
+
+
+## Seven Settings Can Quietly Break Your Search
+
+Start where the 0.01 question gets its answer. Seven collection settings can cap search quality no matter what you tune next. Two examples: a sparse vector missing its IDF modifier stops rare words from counting more than common ones, and a BM25 average length left at the default misjudges every document's length. Neither raises an error. The results are quietly worse.
+
+Your labels decide what you can measure, too. In our runs, 25 labeled queries weren't enough: the noise was wider than any gain our fusion tuning produced. [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/) catches all seven settings and shows how many labeled queries you need before the next four checks are worth trusting.
+
+## Retrieval Delivered, Ranking Buried It
+
+Every search system eventually gets this complaint: a document the user knows exists doesn't come up. They searched the obvious terms, and it landed at rank 40 or nowhere. Either retrieval missed it, or ranking buried it. Those failures need different fixes.
+
+The obvious move is to retrieve more. Retrieving more did help: pushing the candidate limit from 10 to 500 lifted the best achievable score by up to 0.28. The score users saw moved by 0.01 at most, because ranking was burying what retrieval had already found. Even `hnsw_ef`, the knob many teams reach for first, moved the final score by at most 0.0022. [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) shows the check that separates missed retrieval from buried relevance before you pay to fix the wrong one.
+
+## One Constant Flipped the Top Result on 202 of 480 Queries
+
+Hybrid search runs a dense and a sparse query, then merges the two result lists with reciprocal rank fusion. One constant, `k`, decides how much that merge favors each list's top-ranked documents. Qdrant defaults to `k=2`, which puts heavy trust in each list's first pick. Switching to the `k=61` from [the original paper](https://dl.acm.org/doi/10.1145/1571941.1572114) changed which document ranked first for 202 of 480 queries on one of our datasets.
+
+That makes `k` worth testing, but you may not need it at all. DBSF, Qdrant's other fusion method, takes no parameters and beat default RRF on three of five datasets. [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/) shows which method to reach for, and the one count that picks your `k` when you do sweep it.
+
+## The Reranker Got Credit for Tuning We Skipped
+
+You add a cross-encoder, relevance improves, and you ship it. That improvement costs a model forward pass per candidate on every query, for as long as the reranker runs.
+
+Before you pay that cost on every query, tune fusion first. Part of what a reranker appears to buy is fusion tuning you skipped. The best of four rerankers beat Qdrant's out-of-the-box fusion on all five datasets, but against tuned fusion, most of that lift disappeared, and one win became a loss. [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) shows the cheap test that tells you when the model earns its latency.
+
+## The Query That Got 10 Times Slower Over the Weekend
+
+Your p95 looked fine on Friday. The collection grew over the weekend. On Monday, the same query takes ten times longer, with no error and no config change to blame.
+
+The culprit is rescoring. Quantization keeps a compressed copy of your vectors in RAM and rereads the originals to fix compression error. While the originals fit in memory, that reread is nearly free. Once they stop fitting, it hits disk: the same query went from 4.3 ms to 43.4 ms.
+
+So measure quantization at the memory limit you deploy with, because a machine with spare RAM can hide the disk-read cost. And think twice before switching it off: without rescoring, the dense stage found only six in ten of the true nearest neighbors. [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/) has the protocol for choosing between speed and recall, and the signal that warns you before your p95 does.
+
+## Start with the Problem You Have
+
+Each article answers one of these:
+
+- You changed a setting and can't tell whether it helped: [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/)
+- A document you know exists comes back buried or missing: [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/)
+- You use hybrid search with the default fusion settings: [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/)
+- You're deciding whether a reranker would pay for its latency: [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/)
+- Latency jumped after the collection grew: [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/)
+
+If more than one fits, start with the first. It builds the labeled query set the other four checks run on, using the queries product keeps forwarding you as raw material. Run the checks, and tuning stops being guesswork.
diff --git a/qdrant-landing/content/blog/vector-space-day-2026-sf.md b/qdrant-landing/content/blog/vector-space-day-2026-sf.md
index 58d7eff0c..736da8cd4 100644
--- a/qdrant-landing/content/blog/vector-space-day-2026-sf.md
+++ b/qdrant-landing/content/blog/vector-space-day-2026-sf.md
@@ -16,6 +16,13 @@ tags:
## Vector Space Day 2026: Powered by Qdrant
+### Recap
+
+[Watch all the videos](https://qdrant.tech/vector-space-day-sf-26-recap/)
+
+[Read the blog recap](https://qdrant.tech/blog/vector-space-day-2026-recap/)
+
+
### About
We’re hosting our second-ever full-day in-person Vector Space Day (https://luma.com/vsd-sf) on June 11th at The Midway in San Francisco, and you’re invited.
diff --git a/qdrant-landing/content/blog/vsd26-post-event.md b/qdrant-landing/content/blog/vsd26-post-event.md
index 710c99407..b5145e7f4 100644
--- a/qdrant-landing/content/blog/vsd26-post-event.md
+++ b/qdrant-landing/content/blog/vsd26-post-event.md
@@ -14,6 +14,10 @@ tags:
- blog
---
+### Recap
+
+[Watch all the videos](https://qdrant.tech/vector-space-day-sf-26-recap/)
+
On June 11th, 2026, over 350 developers, researchers, and engineers came together at The Midway in San Francisco for **Vector Space Day**, our first event of its kind in the United States and our first major gathering in San Francisco.
This was a single day, single stage, across three tracks: Agents and Memory, Search and Retrieval, and Edge and Robotics. Hosted by our MC for the day, [Adam Chan](https://www.linkedin.com/in/itsajchan/), who kept the energy flowing from opening keynotes to the final hackathon reveal.
diff --git a/qdrant-landing/content/blog/what-is-vector-similarity.md b/qdrant-landing/content/blog/what-is-vector-similarity.md
index 4f52fa48a..c49d3d809 100644
--- a/qdrant-landing/content/blog/what-is-vector-similarity.md
+++ b/qdrant-landing/content/blog/what-is-vector-similarity.md
@@ -145,7 +145,7 @@ The vector index in Qdrant employs the Hierarchical Navigable Small World (HNSW)
### Scalability
-For massive datasets and demanding workloads, Qdrant supports [distributed deployment](/documentation/distributed_deployment/) from v0.8.0. In this mode, you can set up a Qdrant cluster and distribute data across multiple nodes, enabling you to maintain high performance and availability even under increased workloads. Clusters support sharding and replication, and harness the Raft consensus algorithm to manage node coordination.
+For massive datasets and demanding workloads, Qdrant supports [distributed deployment](/documentation/scaling/distributed_deployment/) from v0.8.0. In this mode, you can set up a Qdrant cluster and distribute data across multiple nodes, enabling you to maintain high performance and availability even under increased workloads. Clusters support sharding and replication, and harness the Raft consensus algorithm to manage node coordination.
Qdrant also supports vector [quantization](/documentation/manage-data/quantization/) to reduce memory footprint and speed up vector similarity searches, making it very effective for large-scale applications where efficient resource management is critical.
diff --git a/qdrant-landing/content/course/_index.md b/qdrant-landing/content/course/_index.md
index 19d446f4b..56b7b0c79 100644
--- a/qdrant-landing/content/course/_index.md
+++ b/qdrant-landing/content/course/_index.md
@@ -15,6 +15,23 @@ Whether you’re new to Qdrant or building production-grade systems, our guided
## Available Now
+{{< course-card
+ title="Qdrant Beginner Course"
+ image="/icons/outline/training-white.svg"
+ link="/course/beginners/"
+>}}
+**What you'll gain:**
+- Why Traditional Search Falls Short
+- Embeddings and Distance Metrics
+- Vector Search First Principles
+- Sparse, Dense, and Hybrid Search
+- Designing a Vector Search System
+- Capstone: Multimodal Supplier Risk Intelligence
+
+Time to Complete: under 5 hours
+Includes: videos, code notebooks, projects, certification
+{{< /course-card >}}
+
{{< course-card
title="Qdrant Essentials Course"
image="/icons/outline/rocket-white-light.svg"
@@ -51,27 +68,6 @@ Includes: videos, code notebooks, projects, certification
## Upcoming Courses
-### Beginner Level
-Beginner courses require no previous experience with Qdrant and are useful for building a strong foundation in vector search.
-
-{{< accordion >}}
-- title: "Qdrant Fundamentals"
- content: |
- - Vector Search Concepts
- - Setting Up Qdrant
- - Creating and Managing Collections
- - Ingesting Vector Embeddings
- - Running Your First Query
-
-
- Time to Complete: 2 hours (TBD)
- Includes: videos, code notebooks
-
-
-
-
-{{< /accordion >}}
-
### Intermediate Level
Intermediate courses are recommended for those that have completed the Beginner Level Modules first, and extend knowledge into more practical usage of Qdrant in the real-world.
@@ -148,4 +144,4 @@ Advanced courses are recommended for those that have completed the Beginner and
{{< /accordion >}}
-**Want something not mentioned above? Email [devrel@qdrant.com](emailto:devrel@qdrant.com) and let us know!**
\ No newline at end of file
+**Want something not mentioned above? Email [devrel@qdrant.com](mailto:devrel@qdrant.com) and let us know!**
diff --git a/qdrant-landing/content/course/beginners/_index.md b/qdrant-landing/content/course/beginners/_index.md
new file mode 100644
index 000000000..dfa9b4b35
--- /dev/null
+++ b/qdrant-landing/content/course/beginners/_index.md
@@ -0,0 +1,206 @@
+---
+title: "Beginner Course"
+page_title: "Qdrant Beginner Course"
+short_description: "Learn the fundamentals of vector search: why keyword search struggles, how semantic search improves it, embeddings, distance metrics, and hybrid systems."
+description: "Understand why traditional search struggles and how modern semantic search improves it, and build your first search system."
+content:
+ sidebarTitle: "Beginner Course"
+ menuTitle:
+ text: Course Overview
+ url: /course/beginners/
+ nextButton: Continue to Next Step
+ nextDay: Complete
+ title: "Beginner Course"
+ description: "Understand why traditional search struggles and how modern semantic search improves it, and build your first search system."
+partition: course
+---
+
+# Beginner Course
+
+**Learn the fundamentals of vector search**
+
+Understand why traditional search struggles and how modern semantic search improves it. Learn about embeddings, distance metrics, and hybrid search systems.
+
+
+
+
+
+
+
+{{< cards-list >}}
+- icon: /icons/outline/play-white.svg
+ title: 6 modules
+ content: From setting up dependencies to a hands-on capstone project
+- icon: /icons/outline/cloud-check-blue.svg
+ title: Shareable certificate
+ content: Earn a digital certificate upon completion
+- icon: /icons/outline/time-blue.svg
+ title: Flexible schedule
+ content: Learn at your own pace
+- icon: /icons/outline/plan.svg
+ title: Beginner level
+ content: No prior experience required
+
+{{< /cards-list >}}
+
+
+
+## What You'll Learn
+{{< course-card
+ title="Skills you'll gain:"
+ image="/icons/outline/training-white.svg"
+ type="wide-list">}}
+
+- Why traditional search struggles and how modern semantic search improves it
+- How embeddings convert text to vectors that capture meaning
+- Distance metrics: cosine similarity, dot product, Euclidean and Manhattan
+- Hybrid search: combining dense and sparse retrieval
+- Building your first Qdrant collection and queries
+
+{{< /course-card >}}
+
+### The Path
+
+**Module 0**: Setting Up Dependencies. Configure your environment and get started with the basics.
+
+**Module 1**: Let's Understand Search. Understand why traditional search struggles and how modern semantic search improves it.
+
+**Module 2**: First Principles of Vector Search. Anatomy of a vector - how data is stored, indexed, and retrieved in Qdrant.
+
+**Module 3**: Sparse vs Dense vs Hybrid Search. Understand dense vs sparse search, when each fails, and how hybrid systems combine them.
+
+**Module 4**: Designing a Vector Search System. How to design a vector search system - layers, filtering, RAG, and deployment.
+
+**Module 5**: Capstone - Multimodal Supplier Risk Intelligence. Ingest, cluster, and query multimodal supplier signals across languages.
+
+**Bonus Module**: Further Reading. A roundup of advanced techniques for further reading: score boosting, relevance feedback, MMR, and re-ranking.
+
+## How the Course Works
+
+{{< cards-list >}}
+
+- icon: /icons/outline/training-purple.svg
+ title: Bite-sized lessons
+ content: Short, friendly modules you can finish in one sitting
+- icon: /icons/outline/hacker-purple.svg
+ title: Learn by doing
+ content: Follow along with real examples and hands-on exercises
+- icon: /icons/outline/similarity-blue.svg
+ title: One step at a time
+ content: Each module builds on the last, so nothing feels out of reach
+- icon: /icons/outline/copy.svg
+ title: Go at your own pace
+ content: Pause anytime and pick up right where you left off
+ {{< /cards-list >}}
+
+
+
+## Syllabus
+
+{{< accordion >}}
+- title: "Module 0: Setting Up Dependencies"
+ content: |
+ - Qdrant Cloud Setup
+ - Implementing a Basic Vector Search
+
+
+
+
+- title: "Module 2: First Principles of Vector Search"
+ content: |
+ - What is a Vector?
+ - How Dimensions Represent Meaning
+ - Similarity Under the Hood
+ - Your First Qdrant Collection
+ - Points, Payloads, and Queries
+
+
+
+{{< /accordion >}}
+
+## Who It's For
+
+Anyone new to vector search who wants to understand the fundamentals. No prior experience with Qdrant or vector search engines required.
+
+## Time Commitment
+
+- Core course (Modules 0-4): under 2 hours
+- Capstone project (Module 5): ~3 hours
+- **Total: under 5 hours**
+- Bonus module: optional, not included in the total above
+- Self-paced, flexible schedule
+
+
+{{< course-card
+ title="Ready to start your vector search journey?"
+ image="/icons/outline/rocket-white-light.svg"
+ link="/course/beginners/module-0/">}}
+**What you'll get**
+- Understand the fundamentals of vector search
+- Learn why semantic search outperforms keyword search
+- Build your first Qdrant collection
+- Foundation for advanced courses
+{{< /course-card >}}
diff --git a/qdrant-landing/content/course/beginners/certification/_index.md b/qdrant-landing/content/course/beginners/certification/_index.md
new file mode 100644
index 000000000..fcef487a2
--- /dev/null
+++ b/qdrant-landing/content/course/beginners/certification/_index.md
@@ -0,0 +1,26 @@
+---
+title: "Qdrant Beginner Certification"
+short_description: "Validate your vector search fundamentals with an official certification exam covering semantic search, embeddings, and hybrid retrieval."
+description: "Earn the official Qdrant Beginner certification: prove you can build collections, choose distance metrics, filter payloads, and run hybrid search."
+isLesson: true
+weight: 100
+---
+
+# Qdrant Beginner Certification
+
+Congratulations! You've completed the **Qdrant Beginner** course.
+
+Along the way you learned why keyword search falls short, how embeddings capture meaning, how distance metrics compare that meaning, and how hybrid search brings dense and sparse retrieval together. That effort deserves professional recognition.
+
+## 🏆 Get #QdrantCertified
+
+You've got the fundamentals down, and now it's time to prove it. Validate your skills with our official certification.
+
+**Head over to [train.qdrant.dev](https://train.qdrant.dev) to take the exam.**
+
+Passing this exam shows you can:
+
+* **Explain** when and why semantic search beats keyword search.
+* **Turn** text into embeddings and choose the right distance metric for the job.
+* **Build** a Qdrant collection and run vector, filtered, and hybrid queries.
+* **Design** a complete search system end to end, all the way to a multimodal capstone project.
diff --git a/qdrant-landing/content/course/beginners/module-0/_index.md b/qdrant-landing/content/course/beginners/module-0/_index.md
new file mode 100644
index 000000000..4b819e6bb
--- /dev/null
+++ b/qdrant-landing/content/course/beginners/module-0/_index.md
@@ -0,0 +1,20 @@
+---
+title: "Module 0: Setting Up Dependencies"
+short_description: "Module 0 of the Beginner Course: set up Qdrant Cloud, build a first vector search, and get started with the basics."
+description: "Set up Qdrant and build your first vector search app. Learn how to configure Qdrant Cloud, run a basic search, and get started with the fundamentals."
+isLesson: true
+weight: 10
+---
+
+{{< date >}} Module 0 {{< /date >}}
+
+# Setting Up Dependencies
+
+Get started with Qdrant by setting up your environment and building your first vector search application.
+
+## Today's Path
+
+1. Qdrant Cloud Setup
+2. Implementing a Basic Vector Search
+
+By the end, you'll have a working Qdrant setup and a complete first search running.
diff --git a/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md b/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md
new file mode 100644
index 000000000..efe0034c4
--- /dev/null
+++ b/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md
@@ -0,0 +1,85 @@
+---
+title: "Implementing a Basic Vector Search"
+short_description: "Walk through your first vector search: connect to Qdrant, create a collection, insert points, and run similarity queries with the Python client."
+description: Learn how to build a basic vector search in Qdrant. Create collections, insert vectors, and run your first similarity search step-by-step with Python.
+weight: 3
+isLesson: true
+---
+
+{{< date >}} Module 0 {{< /date >}}
+
+# Implementing a Basic Vector Search
+
+
+
+In this lesson you'll build your very first search, one small step at a time. You'll connect to Qdrant, create a place to store data, add a few example vectors, and then ask Qdrant to find the closest match. Every step has runnable code, so follow along in a notebook or script.
+
+A quick vocabulary note before you start: a **vector** is just a list of numbers that represents something (a piece of text, an image, a product). Searching by vectors means finding the entries whose numbers are closest to your query's numbers. That's the whole idea, and the code below makes it concrete.
+
+## Before You Start
+
+This course requires Python 3.11 or above installed
+
+## Step 1: Install the Qdrant Client
+
+The **client** is the Python library that lets your code talk to Qdrant. Install it first:
+
+```python
+!pip install qdrant-client
+```
+
+## Step 2: Import the Libraries You'll Need
+
+Import two things from the package: `QdrantClient`, which opens the connection, and `models`, which holds the building blocks you'll use to describe collections and points.
+
+```python
+from qdrant_client import QdrantClient, models
+```
+
+## Step 3: Connect to Qdrant Cloud
+
+Use the cluster URL and API key from the previous lesson. If you saved them in a `.env` file, this reads them automatically:
+
+```python
+import os
+
+client = QdrantClient(url=os.getenv("QDRANT_URL"), api_key=os.getenv("QDRANT_API_KEY"))
+
+# For Colab:
+# from google.colab import userdata
+# client = QdrantClient(url=userdata.get("QDRANT_URL"), api_key=userdata.get("QDRANT_API_KEY"))
+```
+
+**Tip:** For quick experiments with no cloud account at all, you can use `client = QdrantClient(":memory:")`. It runs entirely in memory, but your data disappears when the program stops.
+
+## Step 4: Create a Collection
+
+A [collection](/documentation/manage-data/collections/) is where your vectors live. It's a lot like a table in a regular database: a named container for related data. When you create one, you tell Qdrant two things:
+
+- **Size:** how many numbers each vector has.
+- **Distance metric:** how Qdrant measures whether two vectors are "close."
+
+```python
+# Name your collection
+collection_name = "my_first_collection"
+
+# Create it, describing the vectors it will hold
+client.create_collection(
+ collection_name=collection_name,
+ vectors_config=models.VectorParams(
+ size=4, # each vector has 4 numbers
+ distance=models.Distance.COSINE # how we measure closeness
+ )
+)
+```
+
+This returns `True` when it works.
+
+If completed correctly, you will now have an established Qdrant environment for the rest of the course. Later modules will explain collections, points, distance metrics, and more. Keep going to find out more!
+
+ **Congratulations! You've completed Module 0.** 🎉
diff --git a/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md b/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md
new file mode 100644
index 000000000..3c42fa957
--- /dev/null
+++ b/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md
@@ -0,0 +1,174 @@
+---
+title: "Qdrant Setup"
+short_description: "Spin up a managed Qdrant Cloud cluster, generate API keys, and explore the Web UI for collections, points, and cluster monitoring."
+description: Set up your Qdrant Cloud cluster in minutes. Learn to create collections, manage data, access the Web UI, and connect securely from Python.
+weight: 2
+isLesson: true
+---
+
+{{< date >}} Module 0 {{< /date >}}
+
+# Qdrant Setup
+
+
+
+
+
+
+
+Welcome to your first hands-on step. Before you can search anything, you need a place to store your vectors. That's what Qdrant Cloud gives you: a managed Qdrant environment that runs in the cloud, so there's nothing to install and nothing to keep running on your own local machine. It comes with a secure connection, backups, easier updates, and a clean interface you'll use throughout this course.
+
+Don't worry if some terms here are new. You'll set up a cluster, get a key that lets your code talk to it, and run one quick check to confirm it's working. That's the whole goal for this lesson.
+
+## Create Your Cluster
+
+A **cluster** is your personal Qdrant instance in the cloud. Here's how to create one:
+
+1. Sign up at [cloud.qdrant.io](https://cloud.qdrant.io/signup) with email, Google, or GitHub.
+2. Open **Clusters** and select **Create a Free Cluster**. The Free Tier is enough for this whole course, and you won't be asked for a card.
+
+
+
+3. Pick a region close to you or your users. This keeps things fast.
+4. When the cluster is ready, copy the **API key** and store it somewhere safe. An API key is like a password your code uses to prove it's allowed to reach your cluster, so treat it like one. You can always create new keys later from the **API Keys** section on the cluster page.
+
+
+
+## Access the Web UI
+
+The **Web UI** is a dashboard for looking at your data and running searches without writing code. It's the fastest way to see what's happening inside your cluster while you learn.
+
+1. Select **Cluster UI** in the top corner of the cluster page to open the dashboard.
+
+
+
+### What You Can Do in the Web UI
+
+Use the Web UI to manage collections, inspect data, and check how your searches perform.
+
+#### Main Navigation
+
+- **Console:** Run commands against Qdrant right in the browser. Great for testing and seeing responses without writing a program.
+- **Collections:** See and manage all your collections in one place, and track their status, size, and settings at a glance.
+- **Tutorial:** Follow a guided walkthrough with sample data. You create a collection, add vectors, and run a search with live results.
+
+
+
+- **Datasets:** Load ready-made public datasets into your cluster with one click.
+
+#### Inside a Collection
+
+When you open a collection by selecting its name,
+
+
+
+you'll see a detailed view with several tabs. You don't need all of these yet, so here's a plain-language tour you can come back to later:
+
+
+
+- **Points Tab:** Look at, search, and manage your individual data entries. You can view each entry's data, run a quick "find similar" search, or open a graph view of how it connects to its neighbors.
+- **Info Tab:** A health check for the collection. The one field to know for now is `status` — `green` means everything is healthy.
+- **Cluster Tab:** Shows how your data is spread across machines. You'll care about this only once you scale up.
+- **Search Quality Tab:** Measures how accurate your searches are. Useful later, when you start tuning.
+- **Snapshots Tab:** Manage backups of the collection. You can create a [collection snapshot](/documentation/snapshots/), restore it, or move it to another cluster.
+- **Visualize Tab:** See your vectors as a 2D map. A nice way to build intuition once you have real data loaded.
+- **Graph Tab:** Explore how points connect to their nearest neighbors.
+
+## Connect from Python
+
+Now let's connect from code. First, store your credentials in a file named `.env` at the root of your project (or set them in Colab). Keeping them in a separate file means you won't accidentally paste your key into shared code:
+
+```env
+QDRANT_URL=https://YOUR-CLUSTER.cloud.qdrant.io:6333
+QDRANT_API_KEY=YOUR_API_KEY
+```
+
+Then load those values and create a client. The **client** is the object your Python code uses to send requests to Qdrant:
+
+```python
+from qdrant_client import QdrantClient, models
+import os
+
+client = QdrantClient(url=os.getenv("QDRANT_URL"), api_key=os.getenv("QDRANT_API_KEY"))
+
+# For Colab:
+# from google.colab import userdata
+# client = QdrantClient(url=userdata.get("QDRANT_URL"), api_key=userdata.get("QDRANT_API_KEY"))
+
+# Quick health check
+collections = client.get_collections()
+print(f"Connected to Qdrant Cloud: {len(collections.collections)} collections")
+```
+
+If that prints a line about being connected, you're done. That's the whole setup.
+
+## Other Ways to Connect
+
+You can also reach your cluster directly over the web, without Python. This is handy for a quick test:
+
+```bash
+# Using the api-key header
+curl -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333/collections \
+ --header 'api-key: '
+
+# Using the Authorization header
+curl -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333/collections \
+ --header 'Authorization: Bearer '
+```
+
+## Quick Validation
+
+If you want to double-check the connection, these two commands confirm your cluster is up and reachable:
+
+```bash
+# Service health
+curl -s "$QDRANT_URL/healthz" -H "api-key: $QDRANT_API_KEY"
+
+# List collections
+curl -s "$QDRANT_URL/collections" -H "api-key: $QDRANT_API_KEY"
+```
+
+## Good Practices
+
+A few habits worth starting now:
+
+- Keep your key out of your code. Use an environment variable or a secrets manager.
+- Rotate your API keys now and then from the cluster **Access** tab.
+- Use HTTPS only, and tighten access before you expose a cluster to the public internet.
+
+## Common Issues
+
+- **Authentication error:** Recheck the API key and the `api-key` header. A stray space or a missing character is the usual cause.
+- **Connection error:** Confirm the cluster is running and the region URL is correct. Some workplace networks block outbound connections, so try from a personal network if a request hangs.
+
+## Qdrant Cloud Inference
+
+This part is optional, but good to know it exists. Normally you turn text or images into vectors yourself before storing them. **[Cloud Inference](/cloud-inference/)** does that step for you inside Qdrant Cloud: you send raw text or images, and Qdrant creates the vectors and stores them in one call. You'll create vectors by hand in the next lessons so you understand what's happening, but this is a shortcut you can reach for later.
+
+
+
+
+
+Learn more in the [Qdrant Cloud Inference documentation](/documentation/cloud/inference/).
+
+## Qdrant Agent Skills
+
+If you're using an AI coding assistant (Claude Code, Cursor, and others) alongside this course, install the [Qdrant Advisor skill](https://qdrant.tech/documentation/skills/#the-qdrant-advisor) early with this simple command:
+
+```bash
+npx skills add qdrant/skills/meta/qdrant-advisor
+```
+
+It's a single assistant that can troubleshoot and advise on any Qdrant deployment: when you describe a problem like slow search, memory climbing toward an out-of-memory crash, a stuck optimizer, a scaling decision, it searches live documentation, pulls only the branch of guidance that matches your symptom, and grounds its diagnosis in that current, official guidance instead of stale training data.
diff --git a/qdrant-landing/content/course/essentials/_index.md b/qdrant-landing/content/course/essentials/_index.md
index 51825162e..c81a20df1 100644
--- a/qdrant-landing/content/course/essentials/_index.md
+++ b/qdrant-landing/content/course/essentials/_index.md
@@ -3,6 +3,7 @@ title: "Qdrant Essentials Course"
page_title: Qdrant Essentials Course
short_description: "Build production vector search skills in seven days: hybrid retrieval, multivector reranking, quantization, sharding, and multitenancy."
description: Learn hybrid search, multivectors, and production deployment in 7 days. Build and ship a docs search engine.
+weight: 10
content:
sidebarTitle: Qdrant Essentials
menuTitle:
@@ -173,7 +174,7 @@ Build the vector search skills that matter: hybrid retrieval, multivector rerank
content: |
- AI & LLM Frameworks (Haystack, Jina AI, TwelveLabs)
- Data Processing (Unstructured.io)
- - ML Platforms & Analytics (Tensorlake, Vectorize.io, Superlinked, Quotient)
+ - ML Platforms & Analytics (Tensorlake, Vectorize.io, Superlinked)